{
  "benchmark_uses": [
    {
      "benchmark_id": "benchmark-dual-human-crispr-cas9-library",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-08-02",
          "id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-benchmark-dual-human-crispr-cas9-library-3-use-evidence-1",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "874d21c535461e8fe020d585de0c412948736c9c14c902474c8ec11f69de00c8",
            "type": "section",
            "value": "Results and discussion, Par14"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-benchmark-dual-human-crispr-cas9-library-3-use-evidence-2",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "874d21c535461e8fe020d585de0c412948736c9c14c902474c8ec11f69de00c8",
            "type": "section",
            "value": "Results and discussion, Par14"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-benchmark-dual-human-crispr-cas9-library-3-use-evidence-3",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "4ec5530e16f32eafa83fae6bd99781cc94eaf4b23e79700435ec0b1ee51654e6",
            "type": "section",
            "value": "Methods, Sec9, Par26"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-benchmark-dual-human-crispr-cas9-library-3-use-evidence-4",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "dd0f038f47f415483864de293af96c0de7540b456bce03c6286efe1c2eeee2f3",
            "type": "section",
            "value": "Methods, Sec9, Par26"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        }
      ],
      "id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-benchmark-dual-human-crispr-cas9-library-3-use",
      "metric_labels": [],
      "model_ids": [],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "benchmark-creation",
      "reporting_gaps": [
        "No benchmark version is reported.",
        "The final neutral-gene count is not printed after eight genes were removed.",
        "The released dataset's license and version are not reported."
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_applicable",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-applicable",
        "type": "unknown"
      },
      "status": "non-evaluation",
      "verification": {
        "last_verified": "2026-08-02",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
      "work_version_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-pmc-version-1"
    },
    {
      "benchmark_id": "benchmark-dual-human-crispr-cas9-library",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-08-02",
          "id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-benchmark-dual-human-crispr-cas9-library-4-use-evidence-1",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "b6cffa5ad95ff4541993e85c39b898d62ef30574dce11ca692b18ff7c47d9b82",
            "type": "section",
            "value": "Results and discussion, Par14"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-benchmark-dual-human-crispr-cas9-library-4-use-evidence-2",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "66a116f9e5a267c0aa4ef9d499b38d61031f2e76d31b8d9cbc06acba3a5b5e61",
            "type": "section",
            "value": "Results and discussion, Par14"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-benchmark-dual-human-crispr-cas9-library-4-use-evidence-3",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "b063c6d1007272161a3d6f21c1f64fd8deac7322c00976c923a05b294c470c12",
            "type": "section",
            "value": "Results and discussion, Par14"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-benchmark-dual-human-crispr-cas9-library-4-use-evidence-4",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "dccbcb43398754742b5b877360dfbf891f2d1cb6dec02d479cf8bac1a30f0bee",
            "type": "section",
            "value": "Results and discussion, Par14"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        }
      ],
      "id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-benchmark-dual-human-crispr-cas9-library-4-use",
      "metric_labels": [
        "log-fold changes",
        "Chronos gene fitness estimate"
      ],
      "model_ids": [],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "evaluation",
      "reporting_gaps": [
        "The benchmark version is not reported.",
        "The sample size and uncertainty for the reported −0.9 delta are not reported.",
        "Other performance values appear only in figures; no rasterized figure pages were supplied.",
        "Run-level seeds are not reported.",
        "Conflicted metric claim omitted after independent verification; the evaluation relationship is published conservatively.",
        "benchmark version",
        "realized n/scope",
        "exact model",
        "numeric result"
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_reported",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-reported",
        "type": "unknown"
      },
      "status": "partial",
      "verification": {
        "last_verified": "2026-08-02",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
      "work_version_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-pmc-version-1"
    },
    {
      "benchmark_id": "benchmark-human-crispr-cas9-library",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-08-02",
          "id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-benchmark-human-crispr-cas9-library-1-use-evidence-1",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "652b34a951af118b76a7a8e39b41ea6acf996813a8b7bcaa20e093732222e8a8",
            "type": "section",
            "value": "Results and discussion, Par11"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-benchmark-human-crispr-cas9-library-1-use-evidence-2",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "24e393d6f1c90f63001895a1ef5f8c5ef57d6e391c7ca753e172580946d53eb7",
            "type": "section",
            "value": "Results and discussion, Par11"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-benchmark-human-crispr-cas9-library-1-use-evidence-3",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "2da24a2f514c4e793058a6cce4a9b95fc66247d6c88b45347381d2a203fc4ad0",
            "type": "section",
            "value": "Results and discussion, Par11"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-benchmark-human-crispr-cas9-library-1-use-evidence-4",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "ffea2b5ef4d935e0fe0f989c02f677bc4a138fae66c629e787907dfb0e969875",
            "type": "section",
            "value": "Methods, Library designs, Par19"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        }
      ],
      "id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-benchmark-human-crispr-cas9-library-1-use",
      "metric_labels": [],
      "model_ids": [],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "benchmark-creation",
      "reporting_gaps": [
        "The complete original gRNA inventory is not reported.",
        "No benchmark version is reported.",
        "The released dataset's license and version are not reported."
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_applicable",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-applicable",
        "type": "unknown"
      },
      "status": "non-evaluation",
      "verification": {
        "last_verified": "2026-08-02",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
      "work_version_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-pmc-version-1"
    },
    {
      "benchmark_id": "benchmark-human-crispr-cas9-library",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-08-02",
          "id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-benchmark-human-crispr-cas9-library-2-use-evidence-1",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "e7bb47177688072e50e9396dc61837e4fd1b8fab45a57471e344b5e5d7a4590f",
            "type": "section",
            "value": "Results and discussion, Par11"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-benchmark-human-crispr-cas9-library-2-use-evidence-2",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "32cf7b5805c58f912dad086e7160b96c4a35c6de31ffe845a7b173e7afdbf3c0",
            "type": "section",
            "value": "Results and discussion, Par11"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-benchmark-human-crispr-cas9-library-2-use-evidence-3",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "40834315d96c719c4c28fe972f3ba0f411730e72ab02e3a5a8ae7bf40485ac72",
            "type": "figure",
            "value": "Figure 1 caption, panels a and c"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-benchmark-human-crispr-cas9-library-2-use-evidence-4",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "675036106294d8cdfccf390b0d70521e3e20b600b22ead13861d18544cc88401",
            "type": "figure",
            "value": "Figure 1 caption, panel b"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-benchmark-human-crispr-cas9-library-2-use-evidence-5",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "61cd786282ca39d972bda3861b24bde048baec3813defa0d8659d00f3e282dc5",
            "type": "figure",
            "value": "Figure 1 caption, panel c"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-benchmark-human-crispr-cas9-library-2-use-evidence-6",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "019311040e6fabfe4fcac0f6d0974354f9731d923c72c1c620f6186b86cd6694",
            "type": "figure",
            "value": "Figure 1 caption, panel b"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-benchmark-human-crispr-cas9-library-2-use-evidence-7",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "ae844688b48e7d6ed5f855dff86ee77e4905ab7baf991ebdf480cfa25a6569dd",
            "type": "section",
            "value": "Results and discussion, Par13"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-benchmark-human-crispr-cas9-library-2-use-evidence-8",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "d4a43a182e0756ead1b038afbb5d80c7f8b1dfbb18ab65b18255480689763892",
            "type": "section",
            "value": "Results and discussion, Par13"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        }
      ],
      "id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-benchmark-human-crispr-cas9-library-2-use",
      "metric_labels": [
        "log2 fold-change",
        "Chronos gene fitness estimate",
        "Spearman’s rho",
        "Wilcoxon two-sample test P-value",
        "genes targeted by both MinLib guides",
        "genes with at least one corresponding guide in MinLib"
      ],
      "model_ids": [],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "evaluation",
      "reporting_gaps": [
        "The benchmark version and realized gRNA count are not reported.",
        "Most performance values appear only in figures; no rasterized figure pages were supplied.",
        "Run-level seeds and uncertainty intervals are not reported.",
        "benchmark version",
        "realized n/scope",
        "exact model",
        "numeric result"
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_reported",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-reported",
        "type": "unknown"
      },
      "status": "partial",
      "verification": {
        "last_verified": "2026-08-02",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
      "work_version_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-pmc-version-1"
    },
    {
      "benchmark_id": "ab-bind",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-08-14",
          "id": "ab-bind-antibody-binding-mutational-database-for-compu-ab-bind-1-use-evidence-1",
          "locator": {
            "document_page": 3,
            "note": null,
            "printed_page": "395",
            "source_fragment_sha256": "d98519e4515201b9fdd298459f5a3d4e8fefb2d0342119e93244f43f24baabf9",
            "type": "section",
            "value": "Results > AB-Bind database"
          },
          "source_id": "ab-bind-antibody-binding-mutational-database-for-compu",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-08-14",
          "id": "ab-bind-antibody-binding-mutational-database-for-compu-ab-bind-1-use-evidence-2",
          "locator": {
            "document_page": 1,
            "note": null,
            "printed_page": "393",
            "source_fragment_sha256": "5b9f01e36745fe232f491417c34fff66a53682268df5bbba73b580b0fbdac323",
            "type": "section",
            "value": "Abstract"
          },
          "source_id": "ab-bind-antibody-binding-mutational-database-for-compu",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        }
      ],
      "id": "ab-bind-antibody-binding-mutational-database-for-compu-ab-bind-1-use",
      "metric_labels": [],
      "model_ids": [],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "benchmark-creation",
      "reporting_gaps": [
        "No explicit AB-Bind version identifier is reported.",
        "The official repository metadata reports no license.",
        "The source does not reconcile the conflicting broad multiple-mutation count with its detailed partition."
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_applicable",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-applicable",
        "type": "unknown"
      },
      "status": "non-evaluation",
      "verification": {
        "last_verified": "2026-08-14",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "ab-bind-antibody-binding-mutational-database-for-compu",
      "work_version_id": "ab-bind-antibody-binding-mutational-database-for-compu-2015-11-06"
    },
    {
      "benchmark_id": "ab-bind",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-08-14",
          "id": "ab-bind-antibody-binding-mutational-database-for-compu-ab-bind-2-use-evidence-1",
          "locator": {
            "document_page": 5,
            "note": null,
            "printed_page": "397",
            "source_fragment_sha256": "558295ad848528a58b40a39d0d8f870d2107ccd436ee67c3dbcd5f4521e533dd",
            "type": "section",
            "value": "Results > Scoring potentials"
          },
          "source_id": "ab-bind-antibody-binding-mutational-database-for-compu",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-08-14",
          "id": "ab-bind-antibody-binding-mutational-database-for-compu-ab-bind-2-use-evidence-2",
          "locator": {
            "document_page": 11,
            "note": null,
            "printed_page": "403",
            "source_fragment_sha256": "32c6ef7bcca83fa7959e9d7064c70c72f37cd83c82110c00ace00c42999309f8",
            "type": "section",
            "value": "Discussion"
          },
          "source_id": "ab-bind-antibody-binding-mutational-database-for-compu",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        },
        {
          "accessed_date": "2026-08-14",
          "id": "ab-bind-antibody-binding-mutational-database-for-compu-ab-bind-2-use-evidence-3",
          "locator": {
            "document_page": 6,
            "note": null,
            "printed_page": "398",
            "source_fragment_sha256": "79cc7ad36b888e5de07956b008d814f88261f2b45bc7edb745d1d6862f422965",
            "type": "section",
            "value": "Results > Scoring potentials"
          },
          "source_id": "ab-bind-antibody-binding-mutational-database-for-compu",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-14",
          "id": "ab-bind-antibody-binding-mutational-database-for-compu-ab-bind-2-use-evidence-4",
          "locator": {
            "document_page": 6,
            "note": null,
            "printed_page": "398",
            "source_fragment_sha256": "bfa222c4c53607fe4ccb5303ad7ee6d4022ef5696b390a76f77481fae3d9ffb6",
            "type": "section",
            "value": "Results > Scoring potentials"
          },
          "source_id": "ab-bind-antibody-binding-mutational-database-for-compu",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-14",
          "id": "ab-bind-antibody-binding-mutational-database-for-compu-ab-bind-2-use-evidence-5",
          "locator": {
            "document_page": 6,
            "note": null,
            "printed_page": "398",
            "source_fragment_sha256": "a0a84b95fa938b84deb1efe3ea2ed2ed92bff6a37d275d59ae729ca5a942cacc",
            "type": "section",
            "value": "Results > Scoring potentials"
          },
          "source_id": "ab-bind-antibody-binding-mutational-database-for-compu",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-14",
          "id": "ab-bind-antibody-binding-mutational-database-for-compu-ab-bind-2-use-evidence-6",
          "locator": {
            "document_page": 6,
            "note": null,
            "printed_page": "398",
            "source_fragment_sha256": "3e316a493dff0ed6b8aed05f1d394da64e6ae1e56830db6e348ce773e2183c75",
            "type": "section",
            "value": "Results > Scoring potentials"
          },
          "source_id": "ab-bind-antibody-binding-mutational-database-for-compu",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-14",
          "id": "ab-bind-antibody-binding-mutational-database-for-compu-ab-bind-2-use-evidence-7",
          "locator": {
            "document_page": 13,
            "note": null,
            "printed_page": "405",
            "source_fragment_sha256": "c403840b84d795a30ac97d32042c2e10e7b02688cbec8414271a8b2a19acc89d",
            "type": "section",
            "value": "Materials and Methods > Structure generation"
          },
          "source_id": "ab-bind-antibody-binding-mutational-database-for-compu",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-14",
          "id": "ab-bind-antibody-binding-mutational-database-for-compu-ab-bind-2-use-evidence-8",
          "locator": {
            "document_page": 13,
            "note": null,
            "printed_page": "405",
            "source_fragment_sha256": "b8f6fac0ded5a52515284a2b7e19a30d479cad383f6574207ff699a2ad2638c3",
            "type": "section",
            "value": "Materials and Methods > Structure generation"
          },
          "source_id": "ab-bind-antibody-binding-mutational-database-for-compu",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-14",
          "id": "ab-bind-antibody-binding-mutational-database-for-compu-ab-bind-2-use-evidence-9",
          "locator": {
            "document_page": 13,
            "note": null,
            "printed_page": "405, 409",
            "source_fragment_sha256": "d8bf0ca041dae47e659160b961264171174223f20793e90bb5d54bac409f5234",
            "type": "other",
            "value": "Document pages 13 and 17, Structure generation and Reference 61"
          },
          "source_id": "ab-bind-antibody-binding-mutational-database-for-compu",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        }
      ],
      "id": "ab-bind-antibody-binding-mutational-database-for-compu-ab-bind-2-use",
      "metric_labels": [],
      "model_ids": [
        "accelrys-software-inc-discovery-studio",
        "not-reported-ddfire",
        "not-reported-dfire",
        "not-reported-foldx",
        "not-reported-rosetta",
        "not-reported-statium",
        "sirin-et-al-basa"
      ],
      "notes": "Owner-reviewed conservative publication: the creator evaluation is retained only as a partial relationship; conflicted settings and outcomes are omitted pending manual reconciliation.",
      "relation_type": "evaluation",
      "reporting_gaps": [
        "No random seed is reported.",
        "Realized n is not printed for the low- and medium-confidence ROC subsets.",
        "Several method-specific Pearson correlations and confidence intervals appear only in referenced Supporting Information.",
        "Provider identities are not printed for several scoring methods.",
        "benchmark version",
        "realized n/scope",
        "metric",
        "numeric result",
        "prompt and tools",
        "grader and repeats"
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_reported",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-reported",
        "type": "unknown"
      },
      "status": "partial",
      "verification": {
        "last_verified": "2026-08-14",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "ab-bind-antibody-binding-mutational-database-for-compu",
      "work_version_id": "ab-bind-antibody-binding-mutational-database-for-compu-2015-11-06"
    },
    {
      "benchmark_id": "ab-bind",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-08-14",
          "id": "ab-bind-antibody-binding-mutational-database-for-compu-ab-bind-3-use-evidence-1",
          "locator": {
            "document_page": 12,
            "note": null,
            "printed_page": "404",
            "source_fragment_sha256": "171d105dc19a41066fec6c3ee837a750406a5bd28672df8b4a71fa04212db046",
            "type": "section",
            "value": "Discussion"
          },
          "source_id": "ab-bind-antibody-binding-mutational-database-for-compu",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-08-14",
          "id": "ab-bind-antibody-binding-mutational-database-for-compu-ab-bind-3-use-evidence-2",
          "locator": {
            "document_page": 12,
            "note": null,
            "printed_page": "404",
            "source_fragment_sha256": "b3f167b27c1a1aa86b49ed81f1b82ef4eb1ef653aaaeba0502e3c1896ab3d7ef",
            "type": "section",
            "value": "Discussion"
          },
          "source_id": "ab-bind-antibody-binding-mutational-database-for-compu",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        },
        {
          "accessed_date": "2026-08-14",
          "id": "ab-bind-antibody-binding-mutational-database-for-compu-ab-bind-3-use-evidence-3",
          "locator": {
            "document_page": 12,
            "note": null,
            "printed_page": "404–405",
            "source_fragment_sha256": "c8ac1aae88e92450624d0a2a7191079a283456243adf455444842fef1940a108",
            "type": "other",
            "value": "Document pages 12–13, Discussion and Structure generation"
          },
          "source_id": "ab-bind-antibody-binding-mutational-database-for-compu",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-14",
          "id": "ab-bind-antibody-binding-mutational-database-for-compu-ab-bind-3-use-evidence-4",
          "locator": {
            "document_page": 12,
            "note": null,
            "printed_page": "404",
            "source_fragment_sha256": "5f197115563ae90270229515efc08610afc5b789c4bd0158a6ffc009c1bdeb1c",
            "type": "section",
            "value": "Discussion"
          },
          "source_id": "ab-bind-antibody-binding-mutational-database-for-compu",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        }
      ],
      "id": "ab-bind-antibody-binding-mutational-database-for-compu-ab-bind-3-use",
      "metric_labels": [],
      "model_ids": [
        "not-reported-foldx"
      ],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "fine-tuning",
      "reporting_gaps": [
        "The fitting split, realized n, regression specification, and numeric results are not reported; the paper says data not shown."
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_applicable",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-applicable",
        "type": "unknown"
      },
      "status": "non-evaluation",
      "verification": {
        "last_verified": "2026-08-14",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "ab-bind-antibody-binding-mutational-database-for-compu",
      "work_version_id": "ab-bind-antibody-binding-mutational-database-for-compu-2015-11-06"
    },
    {
      "benchmark_id": "abbibench",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma-abbibench-1-use-evidence-1",
          "locator": {
            "document_page": 2,
            "note": null,
            "printed_page": "2",
            "source_fragment_sha256": "931e2050e009fce5b11bc5e5aa6267df865a4d4b091cf194c7168afdc43726da",
            "type": "section",
            "value": "Introduction"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma-abbibench-1-use-evidence-2",
          "locator": {
            "document_page": 2,
            "note": null,
            "printed_page": "2",
            "source_fragment_sha256": "0ff2f5f191531361894c3f085ffce52768505bb7e3e03511c24200db9c0f7d61",
            "type": "section",
            "value": "Introduction"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        }
      ],
      "id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma-abbibench-1-use",
      "metric_labels": [],
      "model_ids": [],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "benchmark-creation",
      "reporting_gaps": [
        "A benchmark version and artifact release date are not reported.",
        "The source conflicts on whether AbBiBench contains 14 or 16 binding-affinity assays."
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_applicable",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-applicable",
        "type": "unknown"
      },
      "status": "non-evaluation",
      "verification": {
        "last_verified": "2026-08-20",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
      "work_version_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma-arxiv-v2"
    },
    {
      "benchmark_id": "abbibench",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma-abbibench-2-use-evidence-1",
          "locator": {
            "document_page": 4,
            "note": null,
            "printed_page": "4",
            "source_fragment_sha256": "e26060c4a4748ac0fb3e6c8672020e2dc34b7c693e140d33456a101b52977d32",
            "type": "section",
            "value": "Section 3"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma-abbibench-2-use-evidence-2",
          "locator": {
            "document_page": 4,
            "note": null,
            "printed_page": "4",
            "source_fragment_sha256": "506213c997e4401b3ed26d6629d976840238a9e017d57931e1a649ecd49e0f74",
            "type": "section",
            "value": "Section 3"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma-abbibench-2-use-evidence-3",
          "locator": {
            "document_page": 19,
            "note": null,
            "printed_page": "19",
            "source_fragment_sha256": "a59552c5c31333a8ae5027eb2c2d10c1288a443b4e3284d272cfd320f7198565",
            "type": "table",
            "value": "Table S2"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma-abbibench-2-use-evidence-4",
          "locator": {
            "document_page": 19,
            "note": null,
            "printed_page": "19",
            "source_fragment_sha256": "a64d4b98b7dc2ed32d512989b32eb7d4d0bd3a1d4f6ed7e3786b0842c34ae0c8",
            "type": "table",
            "value": "Table S2"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma-abbibench-2-use-evidence-5",
          "locator": {
            "document_page": 19,
            "note": null,
            "printed_page": "19",
            "source_fragment_sha256": "d2b10de46ac1ebe89e4076ebcc6b7a84f19b48d424e40d39ec853d5cfca75303",
            "type": "table",
            "value": "Table S2"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma-abbibench-2-use-evidence-6",
          "locator": {
            "document_page": 19,
            "note": null,
            "printed_page": "19",
            "source_fragment_sha256": "0fda6896bdc2f0d5f7c3e05ddaa37ab6b27dcf5f431682286e45b2dace6e6603",
            "type": "table",
            "value": "Table S2"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma-abbibench-2-use-evidence-7",
          "locator": {
            "document_page": 19,
            "note": null,
            "printed_page": "19",
            "source_fragment_sha256": "5d5d3e157cf8fa14e6840159f94327409cf00e06264cec7772d24f31f54f89cb",
            "type": "table",
            "value": "Table S2"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma-abbibench-2-use-evidence-8",
          "locator": {
            "document_page": 19,
            "note": null,
            "printed_page": "19",
            "source_fragment_sha256": "b8a1314255054ebfc77dbf7a13ceca121184a60e979bf8902b3a535a1ec14f30",
            "type": "table",
            "value": "Table S2"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma-abbibench-2-use-evidence-9",
          "locator": {
            "document_page": 8,
            "note": null,
            "printed_page": "8",
            "source_fragment_sha256": "90e9a4be47102bb2c2450b4960b5193d3eecfbd8a84781d325ece8d8559db230",
            "type": "figure",
            "value": "Figure 3"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma-abbibench-2-use-evidence-10",
          "locator": {
            "document_page": 19,
            "note": null,
            "printed_page": "19",
            "source_fragment_sha256": "dc1fcdc13831e35bd26143b853ac04b27f18903b73ac01a75e580862d2256462",
            "type": "table",
            "value": "Table S2"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma-abbibench-2-use-evidence-11",
          "locator": {
            "document_page": 19,
            "note": null,
            "printed_page": "19",
            "source_fragment_sha256": "7e7539e51c748222db0fcaa6d2cc9d48ced3e00d4ec73b6d3ac49ae0e83a02a8",
            "type": "other",
            "value": "Table S2 and registry-context.json model proteinmpnn"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma-abbibench-2-use-evidence-12",
          "locator": {
            "document_page": 8,
            "note": null,
            "printed_page": "8",
            "source_fragment_sha256": "a38f33fe4084cc60c174c974f83880b914674f7b31f182b43eedd5378c02cf7d",
            "type": "figure",
            "value": "Figure 3"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma-abbibench-2-use-evidence-13",
          "locator": {
            "document_page": 19,
            "note": null,
            "printed_page": "19",
            "source_fragment_sha256": "dc9699027dc92feb5ddbf2e821ad338edce59f135c4bd8b0cfdd921b0e5800af",
            "type": "table",
            "value": "Table S2"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma-abbibench-2-use-evidence-14",
          "locator": {
            "document_page": 19,
            "note": null,
            "printed_page": "19",
            "source_fragment_sha256": "991bab105eeff787e0ade779a82ca3e4f776406a0e285fc35985fa5b8f8a0fef",
            "type": "table",
            "value": "Table S2"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma-abbibench-2-use-evidence-15",
          "locator": {
            "document_page": 19,
            "note": null,
            "printed_page": "19",
            "source_fragment_sha256": "fa93bcab75f0e54ce3239f06df3ca07a52554a6845c7d1db41feeb552c51932c",
            "type": "table",
            "value": "Table S2"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma-abbibench-2-use-evidence-16",
          "locator": {
            "document_page": 19,
            "note": null,
            "printed_page": "19",
            "source_fragment_sha256": "99244929748b975684d0cef44209f8aef0dcacc4c4bf8b9f5483473b96e20d98",
            "type": "table",
            "value": "Table S2"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma-abbibench-2-use-evidence-17",
          "locator": {
            "document_page": 19,
            "note": null,
            "printed_page": "19",
            "source_fragment_sha256": "51bd78b9d0e8ccf945bee1df9de46926415bf29b3df98e4eb6a7462fca6f11fc",
            "type": "table",
            "value": "Table S2"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma-abbibench-2-use-evidence-18",
          "locator": {
            "document_page": 19,
            "note": null,
            "printed_page": "19",
            "source_fragment_sha256": "4dcec7d3970745a2f6b07711f15ad9552953ba30c3f1cd361d2769c987783b7c",
            "type": "table",
            "value": "Table S2"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma-abbibench-2-use-evidence-19",
          "locator": {
            "document_page": 19,
            "note": null,
            "printed_page": "19",
            "source_fragment_sha256": "48992131fc368d6dd872e9b56d64dfc2d882ecda63462f15086308d998e09b2f",
            "type": "table",
            "value": "Table S2"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        }
      ],
      "id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma-abbibench-2-use",
      "metric_labels": [],
      "model_ids": [
        "esm-if1",
        "not-reported-antiberty",
        "not-reported-antifold",
        "not-reported-currab",
        "not-reported-diffab",
        "not-reported-diffab-fixbb",
        "not-reported-dymean",
        "not-reported-dymean-fixbb",
        "not-reported-esm2",
        "not-reported-esm3",
        "not-reported-mean",
        "not-reported-mean-fixbb",
        "not-reported-progen2-large",
        "not-reported-prosst",
        "not-reported-saprot",
        "proteinmpnn",
        "protgpt2"
      ],
      "notes": "Owner-reviewed conservative publication: the creator evaluation is retained only as a partial relationship; conflicted settings and outcomes are omitted pending manual reconciliation.",
      "relation_type": "evaluation",
      "reporting_gaps": [
        "Exact model versions, providers, seeds, repeats, and confidence intervals are not reported.",
        "The source conflicts on light-chain inclusion and on the 1mlc evaluation count.",
        "Exact model and tool versions, compute time, and confidence intervals are not reported.",
        "DiffAb seeds are described only as reaching up to 15; the exact seed list is absent.",
        "The numeric ELISA detection threshold is not reported.",
        "The final Pareto-candidate count conflicts between 18 and 21.",
        "Exact model versions, seeds, repeats, and exact p-values are not reported.",
        "benchmark version",
        "realized n/scope",
        "metric",
        "numeric result",
        "prompt and tools",
        "grader and repeats"
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_reported",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-reported",
        "type": "unknown"
      },
      "status": "partial",
      "verification": {
        "last_verified": "2026-08-20",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
      "work_version_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma-arxiv-v2"
    },
    {
      "benchmark_id": "cam-benchmark",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-cam-benchmark-5-use-evidence-1",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "3ecef64ce209f99df5aff455e889e05aef794a1468507ef2212deef6f716c976",
            "type": "section",
            "value": "Results — Genetic algorithms can be applied to higher-dimensional design problems"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-cam-benchmark-5-use-evidence-2",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "0f39807c6fb3f3e9eb7495acc5e658fd98f0f29df748527d77bd0d83c6e42b93",
            "type": "section",
            "value": "Methods — Structure preparation"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        }
      ],
      "id": "an-integrative-approach-to-protein-sequence-design-thr-cam-benchmark-5-use",
      "metric_labels": [],
      "model_ids": [],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "benchmark-creation",
      "reporting_gaps": [
        "No formal benchmark version or release identifier is reported.",
        "The repository and benchmark-data license is not reported."
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_applicable",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-applicable",
        "type": "unknown"
      },
      "status": "non-evaluation",
      "verification": {
        "last_verified": "2026-07-31",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "an-integrative-approach-to-protein-sequence-design-thr",
      "work_version_id": "an-integrative-approach-to-protein-sequence-design-thr-pmc-version-1"
    },
    {
      "benchmark_id": "cam-benchmark",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-cam-benchmark-6-use-evidence-1",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "74a63b0f5d7cc3b04b46a3c4687dcbfa5703195f65e2db2daa2a46f0967f40a6",
            "type": "section",
            "value": "Results — GA[AF2Rank] recapitulates the CaM charge profile"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-cam-benchmark-6-use-evidence-2",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "87b872e188012dde7bc79f0050ec56e04f3529823616351d129b2015f528a74b",
            "type": "section",
            "value": "Methods — Structure preparation"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-cam-benchmark-6-use-evidence-3",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "0f39807c6fb3f3e9eb7495acc5e658fd98f0f29df748527d77bd0d83c6e42b93",
            "type": "section",
            "value": "Methods — Structure preparation"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-cam-benchmark-6-use-evidence-4",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "ac12d20d3f984b9203d14f0cc49274850b15975e0c0cb26517eaa0bb0fbe2eec",
            "type": "section",
            "value": "Methods — Structure preparation"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-cam-benchmark-6-use-evidence-5",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "9a88ee81f4e4d696841f42f60c6348bf1df58308780846408c88a5afcc6a69bc",
            "type": "section",
            "value": "Methods — Designable positions"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-cam-benchmark-6-use-evidence-6",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "85b18c782e0732830776dc1204b5fa2da120f71492cf3a8fea580d6090a7db31",
            "type": "section",
            "value": "Methods — Structure preparation; Genetic algorithm; Results — Genetic algorithms can be applied to higher-dimensional design problems"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-cam-benchmark-6-use-evidence-7",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "d0ecae25229cc22728b3965bb6ee3777ff08bdfd609c9590a1f841a4bd207e24",
            "type": "section",
            "value": "Methods — ProteinMPNN; reference 20"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-cam-benchmark-6-use-evidence-8",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "851eca4258667d01fdaeb5ba8407e56df82370828f65a8a8916a6925891f372e",
            "type": "section",
            "value": "Methods — ESM-1v; reference 27"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-cam-benchmark-6-use-evidence-9",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "d880c3bbc4b5b4a55022c6716f9f4b0823b56aec6f1e5c5248a46ee4f6af5fd2",
            "type": "section",
            "value": "Methods — AF2Rank; reference 14"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-cam-benchmark-6-use-evidence-10",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "73e829b7645ff5676e17c9f65b5094b9ba795d985eb2fe19ce7b1d544a9d1aa7",
            "type": "section",
            "value": "Methods — Sequence analysis"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-cam-benchmark-6-use-evidence-11",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "5d711ea53328b2443eff56d2d96a03db3d02386fbcf70a4ee2872e0761c8d645",
            "type": "figure",
            "value": "Fig. 5"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-cam-benchmark-6-use-evidence-12",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "c7f2c62695aac513def7cd891643f022fccae7460e24091f33f8363ba49181ff",
            "type": "section",
            "value": "Methods — ProteinMPNN; Genetic algorithm"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-cam-benchmark-6-use-evidence-13",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "2b79bf4eb79b770acaae96ce97df017cb4191bab24c83ad725fc0ee5bbbc885d",
            "type": "section",
            "value": "Methods — AF2Rank"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-cam-benchmark-6-use-evidence-14",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "e1fa0b58eed5377b2ebec35793bbf02059b5ca42bcde192447309a3761560adc",
            "type": "section",
            "value": "Fig. 5; Methods — Hypervolume"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-cam-benchmark-6-use-evidence-15",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "97bb048b5e59391e0a64f9d04e11f20eefa2f39c01ed8f8425f55cc42f45dd34",
            "type": "section",
            "value": "Fig. 5; Methods — Hypervolume"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-cam-benchmark-6-use-evidence-16",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "7d066b58910a9ae5f69c558631cd6aefab25eaa1b6cd8d5d10f02f0b8eafc9c9",
            "type": "section",
            "value": "Results — GA[AF2Rank] recapitulates the CaM charge profile; Figs. 5 and 3"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        }
      ],
      "id": "an-integrative-approach-to-protein-sequence-design-thr-cam-benchmark-6-use",
      "metric_labels": [
        "native sequence recovery",
        "ESM-1v log likelihood score",
        "pMPNN-SD negative log likelihood score",
        "AF2Rank composite score",
        "pMPNN-SD log likelihood score hypervolume",
        "AF2Rank composite score hypervolume",
        "per-position sequence entropy"
      ],
      "model_ids": [
        "jumper-et-al-alphafold2",
        "meier-et-al-esm-1v",
        "proteinmpnn"
      ],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "evaluation",
      "reporting_gaps": [
        "realized n/scope",
        "Exact primary metric values are not printed in body text or tables; plotted values are unlabeled.",
        "The starting random seed is not reported.",
        "Confidence intervals and statistical uncertainty are not reported.",
        "Exact release dates for the evaluated models are not reported.",
        "benchmark version",
        "numeric result"
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_reported",
        "selection": null,
        "selection_method": "Table 1's fourteen preprocessed CaM structures; NSGA-III with mutation rates 0.1, 0.3, and 0.5 and reference-direction count equal to population size.",
        "subset_id": null,
        "subset_kind": "not-reported",
        "type": "unknown"
      },
      "status": "partial",
      "verification": {
        "last_verified": "2026-07-31",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "an-integrative-approach-to-protein-sequence-design-thr",
      "work_version_id": "an-integrative-approach-to-protein-sequence-design-thr-pmc-version-1"
    },
    {
      "benchmark_id": "papd-benchmark",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-papd-benchmark-3-use-evidence-1",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "34d261bc074e5ace5f53c1cffe5fd9099051baa146e4ac933c44fb16838803dd",
            "type": "section",
            "value": "Results — Genetic algorithms can be applied to higher-dimensional design problems"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-papd-benchmark-3-use-evidence-2",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "428f04090aa0b0d79809a82d27a309df646ca8c1fe0746a2c3f92596d3b7b834",
            "type": "section",
            "value": "Methods — Structure preparation"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        }
      ],
      "id": "an-integrative-approach-to-protein-sequence-design-thr-papd-benchmark-3-use",
      "metric_labels": [],
      "model_ids": [],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "benchmark-creation",
      "reporting_gaps": [
        "No formal benchmark version or release identifier is reported.",
        "The repository and benchmark-data license is not reported."
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_applicable",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-applicable",
        "type": "unknown"
      },
      "status": "non-evaluation",
      "verification": {
        "last_verified": "2026-07-31",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "an-integrative-approach-to-protein-sequence-design-thr",
      "work_version_id": "an-integrative-approach-to-protein-sequence-design-thr-pmc-version-1"
    },
    {
      "benchmark_id": "papd-benchmark",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-papd-benchmark-4-use-evidence-1",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "2fe0968244c544d3c255a70113240993c8827d418da08188e799304c890de9a5",
            "type": "section",
            "value": "Results — Genetic algorithms can be applied to higher-dimensional design problems"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-papd-benchmark-4-use-evidence-2",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "848cd33a2bc2f062912f7f517b0dd4422bf797fb25a4a1a0eea407983c427d30",
            "type": "section",
            "value": "Methods — Structure preparation"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-papd-benchmark-4-use-evidence-3",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "0b42ae97cc6ef794b6ab06131a118bee7962046ac4d0cfe1e8070bfeea533310",
            "type": "section",
            "value": "Results — Genetic algorithms can be applied to higher-dimensional design problems"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-papd-benchmark-4-use-evidence-4",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "27bc4b821ed2b78dc5e376c8565960c12ea9eabd12488d72c616d01b62ed6e54",
            "type": "figure",
            "value": "Fig. 1"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-papd-benchmark-4-use-evidence-5",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "9a53bcbc80decb785fcd0a481a5c46bdeeca5cb58dcf40f062a20af6d81a7cb1",
            "type": "section",
            "value": "Methods — Designable positions"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-papd-benchmark-4-use-evidence-6",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "1223d712c0b01b1fc77c9501b8c1e3d7d58b540fe6c7bf5481d20c1e25269b8e",
            "type": "section",
            "value": "Methods — Structure preparation; Results — Genetic algorithms can be applied to higher-dimensional design problems"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-papd-benchmark-4-use-evidence-7",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "d0ecae25229cc22728b3965bb6ee3777ff08bdfd609c9590a1f841a4bd207e24",
            "type": "section",
            "value": "Methods — ProteinMPNN; reference 20"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-papd-benchmark-4-use-evidence-8",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "851eca4258667d01fdaeb5ba8407e56df82370828f65a8a8916a6925891f372e",
            "type": "section",
            "value": "Methods — ESM-1v; reference 27"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-papd-benchmark-4-use-evidence-9",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "d880c3bbc4b5b4a55022c6716f9f4b0823b56aec6f1e5c5248a46ee4f6af5fd2",
            "type": "section",
            "value": "Methods — AF2Rank; reference 14"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-papd-benchmark-4-use-evidence-10",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "73e829b7645ff5676e17c9f65b5094b9ba795d985eb2fe19ce7b1d544a9d1aa7",
            "type": "section",
            "value": "Methods — Sequence analysis"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-papd-benchmark-4-use-evidence-11",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "5d711ea53328b2443eff56d2d96a03db3d02386fbcf70a4ee2872e0761c8d645",
            "type": "figure",
            "value": "Fig. 5"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-papd-benchmark-4-use-evidence-12",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "c7f2c62695aac513def7cd891643f022fccae7460e24091f33f8363ba49181ff",
            "type": "section",
            "value": "Methods — ProteinMPNN; Genetic algorithm"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-papd-benchmark-4-use-evidence-13",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "2b79bf4eb79b770acaae96ce97df017cb4191bab24c83ad725fc0ee5bbbc885d",
            "type": "section",
            "value": "Methods — AF2Rank"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-papd-benchmark-4-use-evidence-14",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "fde51fe071c4b374e56b94533e5208c4312c2952dab1f7e8c3519ec590900dfb",
            "type": "section",
            "value": "Fig. 5; Methods — Hypervolume"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-papd-benchmark-4-use-evidence-15",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "50d10d56f60bae69864d617c985bfe6773bd341568274a5f78ae3a4624c0e23a",
            "type": "section",
            "value": "Fig. 5; Methods — Hypervolume"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-papd-benchmark-4-use-evidence-16",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "2badf73b85a19d250efbb6e968255adb7b283c98c70aba3d4b799cbe8c6b0168",
            "type": "section",
            "value": "Results — Genetic algorithms can be applied to higher-dimensional design problems; Figs. 5 and 3"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        }
      ],
      "id": "an-integrative-approach-to-protein-sequence-design-thr-papd-benchmark-4-use",
      "metric_labels": [
        "native sequence recovery",
        "ESM-1v log likelihood score",
        "pMPNN-SD negative log likelihood score",
        "AF2Rank composite score",
        "pMPNN-SD log likelihood score hypervolume",
        "AF2Rank composite score hypervolume",
        "per-position sequence entropy"
      ],
      "model_ids": [
        "jumper-et-al-alphafold2",
        "meier-et-al-esm-1v",
        "proteinmpnn"
      ],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "evaluation",
      "reporting_gaps": [
        "realized n/scope",
        "Exact primary metric values are not printed in body text or tables; plotted values are unlabeled.",
        "The starting random seed is not reported.",
        "Confidence intervals and statistical uncertainty are not reported.",
        "Exact release dates for the evaluated models are not reported.",
        "benchmark version",
        "numeric result"
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_reported",
        "selection": null,
        "selection_method": "Preprocessed 1N0L, 1PDK, and 1QPP structures; NSGA-II with mutation rates 0.1, 0.3, and 0.5 and otherwise unchanged RfaH hyperparameters.",
        "subset_id": null,
        "subset_kind": "not-reported",
        "type": "unknown"
      },
      "status": "partial",
      "verification": {
        "last_verified": "2026-07-31",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "an-integrative-approach-to-protein-sequence-design-thr",
      "work_version_id": "an-integrative-approach-to-protein-sequence-design-thr-pmc-version-1"
    },
    {
      "benchmark_id": "rfah-benchmark",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-rfah-benchmark-1-use-evidence-1",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "976f6766a97eaca86974c18e41a70639cbc51c3dd1a5a3940437a202e4ab0e59",
            "type": "section",
            "value": "Introduction; Results — The random resetting mutation operator results in slow convergence"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-rfah-benchmark-1-use-evidence-2",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "5746ab7b9a237bab9f7874e25fb9eb477020c4ef36fcda7c81217f61f535cec8",
            "type": "section",
            "value": "Methods — Hypervolume"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        }
      ],
      "id": "an-integrative-approach-to-protein-sequence-design-thr-rfah-benchmark-1-use",
      "metric_labels": [],
      "model_ids": [],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "benchmark-creation",
      "reporting_gaps": [
        "No formal benchmark version or release identifier is reported.",
        "The repository and benchmark-data license is not reported."
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_applicable",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-applicable",
        "type": "unknown"
      },
      "status": "non-evaluation",
      "verification": {
        "last_verified": "2026-07-31",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "an-integrative-approach-to-protein-sequence-design-thr",
      "work_version_id": "an-integrative-approach-to-protein-sequence-design-thr-pmc-version-1"
    },
    {
      "benchmark_id": "rfah-benchmark",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-rfah-benchmark-2-use-evidence-1",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "b880b790a2cc2afc3dc42f089a482d755602ba5ee6d0b9a5c97a77d0b58d8955",
            "type": "figure",
            "value": "Fig. 2"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-rfah-benchmark-2-use-evidence-2",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "23f0081291069ebf2ea7d4b57d63ada8c425b78f851ffaae421e8f2b1eb8c64f",
            "type": "figure",
            "value": "S2 Fig."
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-rfah-benchmark-2-use-evidence-3",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "82e6ed90c80b7bdf67bac5123d99e20567a95b5c6c3a9ac94d3ccf02c0da886e",
            "type": "section",
            "value": "Results — The random resetting mutation operator results in slow convergence"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-rfah-benchmark-2-use-evidence-4",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "173f32f35e2438186a86683bcb0dead6a2164c9a86705544e5642e19b267663a",
            "type": "figure",
            "value": "Fig. 1"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-rfah-benchmark-2-use-evidence-5",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "1436948f6a443a50960456dcdd3cfb4ed98ecc1932f5a606c47125a78f85fe10",
            "type": "section",
            "value": "Methods — Designable positions"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-rfah-benchmark-2-use-evidence-6",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "8b3cb1e91b33bf600132fed874b5f16e7940fff31ea96e9bfb6eb0b64a983deb",
            "type": "section",
            "value": "Methods — Structure preparation; Genetic algorithm"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-rfah-benchmark-2-use-evidence-7",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "d0ecae25229cc22728b3965bb6ee3777ff08bdfd609c9590a1f841a4bd207e24",
            "type": "section",
            "value": "Methods — ProteinMPNN; reference 20"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-rfah-benchmark-2-use-evidence-8",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "851eca4258667d01fdaeb5ba8407e56df82370828f65a8a8916a6925891f372e",
            "type": "section",
            "value": "Methods — ESM-1v; reference 27"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-rfah-benchmark-2-use-evidence-9",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "d880c3bbc4b5b4a55022c6716f9f4b0823b56aec6f1e5c5248a46ee4f6af5fd2",
            "type": "section",
            "value": "Methods — AF2Rank; reference 14"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-rfah-benchmark-2-use-evidence-10",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "73e829b7645ff5676e17c9f65b5094b9ba795d985eb2fe19ce7b1d544a9d1aa7",
            "type": "section",
            "value": "Methods — Sequence analysis"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-rfah-benchmark-2-use-evidence-11",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "bbe368435b4fe0465e8a74e90eea3b27319ab55dde56e3e75f5430337283c15d",
            "type": "figure",
            "value": "Fig. 2"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-rfah-benchmark-2-use-evidence-12",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "c7f2c62695aac513def7cd891643f022fccae7460e24091f33f8363ba49181ff",
            "type": "section",
            "value": "Methods — ProteinMPNN; Genetic algorithm"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-rfah-benchmark-2-use-evidence-13",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "2b79bf4eb79b770acaae96ce97df017cb4191bab24c83ad725fc0ee5bbbc885d",
            "type": "section",
            "value": "Methods — AF2Rank"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-rfah-benchmark-2-use-evidence-14",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "347bc6993f747157e15292fac4c8bf28f6117b48c5788512e987d04c999f29c7",
            "type": "figure",
            "value": "Fig. 2"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-rfah-benchmark-2-use-evidence-15",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "97912c308c6d6112a65a08fdc920dad4a3c6b938952c798a33a4610e6dcbddd7",
            "type": "figure",
            "value": "Fig. 2"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-rfah-benchmark-2-use-evidence-16",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "87db3fd6ba3a9b18b35bd7e60a0b095a52a09b8a689842359b979ec68f082b4a",
            "type": "figure",
            "value": "Fig. 3; Results — The random resetting mutation operator results in slow convergence"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-rfah-benchmark-2-use-evidence-17",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "686a745bfa8d4938a100a718bf205508ad630591178929f94c706fe8b7b80106",
            "type": "section",
            "value": "Methods — Sequence analysis"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        }
      ],
      "id": "an-integrative-approach-to-protein-sequence-design-thr-rfah-benchmark-2-use",
      "metric_labels": [
        "native sequence recovery",
        "ESM-1v log likelihood score",
        "pMPNN-SD negative log likelihood score",
        "AF2Rank composite score",
        "HV[pMPNN]",
        "HV[AF2Rank]",
        "per-position sequence entropy",
        "normalized BLOSUM62 sequence similarity"
      ],
      "model_ids": [
        "jumper-et-al-alphafold2",
        "meier-et-al-esm-1v",
        "proteinmpnn"
      ],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "evaluation",
      "reporting_gaps": [
        "realized n/scope",
        "Exact primary metric values are not printed in body text or tables; plotted values are unlabeled.",
        "Confidence intervals and statistical uncertainty are not reported.",
        "Exact release dates for the evaluated models are not reported.",
        "benchmark version",
        "numeric result"
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_reported",
        "selection": null,
        "selection_method": "Preprocessed 5OND and the first 2LCL model; NSGA-II with two-point crossover, random/ESM position selection, and pMPNN-AD or uniform mutation.",
        "subset_id": null,
        "subset_kind": "not-reported",
        "type": "unknown"
      },
      "status": "partial",
      "verification": {
        "last_verified": "2026-07-31",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "an-integrative-approach-to-protein-sequence-design-thr",
      "work_version_id": "an-integrative-approach-to-protein-sequence-design-thr-pmc-version-1"
    },
    {
      "benchmark_id": "anthropic-computational-biology",
      "benchmark_version": "reported-2026-01-11",
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [
        "anthropic-computational-biology-delta"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "anthropic-computational-biology-evaluation-use-evidence",
          "locator": {
            "note": "Shows Opus 4.1, Sonnet 4.5, Opus 4.5 and an exact +10.5% annotation.",
            "type": "figure",
            "value": "Evals for key life sciences tasks — Computational biology"
          },
          "source_id": "anthropic-healthcare-life-sciences",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/relation_type",
            "/status",
            "/model_ids",
            "/scope",
            "/metric_labels",
            "/evaluation_run_ids",
            "/reporting_gaps",
            "/notes"
          ]
        }
      ],
      "id": "anthropic-computational-biology-evaluation",
      "metric_labels": [
        "Accuracy delta"
      ],
      "model_ids": [
        "claude-opus-4-1",
        "claude-sonnet-4-5",
        "claude-opus-4-5"
      ],
      "notes": "Only the exact Opus 4.5 improvement annotation is normalized; plotted absolute values are not estimated.",
      "relation_type": "evaluation",
      "reporting_gaps": [
        "task count",
        "task data",
        "grader",
        "prompt",
        "tools",
        "repeats",
        "absolute accuracy"
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_reported",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-reported",
        "type": "unknown"
      },
      "status": "normalized",
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Exact relationship and delta are public while the underlying evaluation remains private.",
        "status": "verified"
      },
      "work_id": "anthropic-healthcare-life-sciences",
      "work_version_id": "anthropic-healthcare-life-sciences-2026-01-11"
    },
    {
      "benchmark_id": "anthropic-key-life-sciences-evals",
      "benchmark_version": "reported-2026-01-11",
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "anthropic-key-life-sciences-creation-use-evidence",
          "locator": {
            "note": "Anthropic page presents the three private internal directions and model trend lines.",
            "type": "figure",
            "value": "Evals for key life sciences tasks"
          },
          "source_id": "anthropic-healthcare-life-sciences",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/relation_type",
            "/status",
            "/scope",
            "/notes"
          ]
        }
      ],
      "id": "anthropic-key-life-sciences-creation",
      "metric_labels": [],
      "model_ids": [],
      "notes": "Anthropic is the creator and evaluator of this private internal suite; child-track evaluation relations are recorded separately.",
      "relation_type": "benchmark-creation",
      "reporting_gaps": [],
      "scope": {
        "n": null,
        "reporting_status": "not_applicable",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-applicable",
        "type": "unknown"
      },
      "status": "non-evaluation",
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Official model-provider page is the only public metadata source for this suite.",
        "status": "verified"
      },
      "work_id": "anthropic-healthcare-life-sciences",
      "work_version_id": "anthropic-healthcare-life-sciences-2026-01-11"
    },
    {
      "benchmark_id": "bixbench",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "anthropic-life-sciences-bixbench-evidence",
          "locator": {
            "note": "Names BixBench and compares Sonnet 4.5 with predecessor Sonnet 4, without settings, metric, or results.",
            "type": "section",
            "value": "Making Claude a better research partner, paragraph 2"
          },
          "source_id": "anthropic-life-sciences",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/relation_type",
            "/status",
            "/model_ids",
            "/scope",
            "/metric_labels",
            "/evaluation_run_ids",
            "/reporting_gaps",
            "/notes"
          ]
        }
      ],
      "id": "anthropic-life-sciences-bixbench",
      "metric_labels": [],
      "model_ids": [
        "claude-sonnet-4-5",
        "claude-sonnet-4"
      ],
      "notes": "Anthropic states that Sonnet 4.5 shows a similar improvement over Sonnet 4 on BixBench, but publishes no score or sufficiently specified evaluation setting. No value is inferred from the wording.",
      "relation_type": "evaluation",
      "reporting_gaps": [
        "benchmark version",
        "full, subset, or track scope",
        "realized n",
        "metric and aggregation",
        "numeric results",
        "prompt, tools, and budget",
        "repeats and grader"
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_reported",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-reported",
        "type": "unknown"
      },
      "status": "partial",
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "The comparison claim is preserved as a partial use rather than a fabricated normalized run.",
        "status": "verified"
      },
      "work_id": "anthropic-life-sciences",
      "work_version_id": "anthropic-life-sciences-2025-10-20"
    },
    {
      "benchmark_id": "anthropic-protein-understanding",
      "benchmark_version": "reported-2026-01-11",
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [
        "anthropic-protein-understanding-delta"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "anthropic-protein-understanding-evaluation-use-evidence",
          "locator": {
            "note": "Shows Opus 4.1, Sonnet 4.5, Opus 4.5 and an exact +10.3% annotation.",
            "type": "figure",
            "value": "Evals for key life sciences tasks — Protein understanding"
          },
          "source_id": "anthropic-healthcare-life-sciences",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/relation_type",
            "/status",
            "/model_ids",
            "/scope",
            "/metric_labels",
            "/evaluation_run_ids",
            "/reporting_gaps",
            "/notes"
          ]
        }
      ],
      "id": "anthropic-protein-understanding-evaluation",
      "metric_labels": [
        "Accuracy delta"
      ],
      "model_ids": [
        "claude-opus-4-1",
        "claude-sonnet-4-5",
        "claude-opus-4-5"
      ],
      "notes": "Only the exact Opus 4.5 improvement annotation is normalized; plotted absolute values are not estimated.",
      "relation_type": "evaluation",
      "reporting_gaps": [
        "task count",
        "task data",
        "grader",
        "prompt",
        "tools",
        "repeats",
        "absolute accuracy"
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_reported",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-reported",
        "type": "unknown"
      },
      "status": "normalized",
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Exact relationship and delta are public while the underlying evaluation remains private.",
        "status": "verified"
      },
      "work_id": "anthropic-healthcare-life-sciences",
      "work_version_id": "anthropic-healthcare-life-sciences-2026-01-11"
    },
    {
      "benchmark_id": "anthropic-scientific-figure-interpretation",
      "benchmark_version": "reported-2026-01-11",
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [
        "anthropic-scientific-figure-delta"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "anthropic-scientific-figure-evaluation-use-evidence",
          "locator": {
            "note": "Shows Opus 4.1, Sonnet 4.5, Opus 4.5 and an exact +13.2% annotation.",
            "type": "figure",
            "value": "Evals for key life sciences tasks — Scientific figure interpretation"
          },
          "source_id": "anthropic-healthcare-life-sciences",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/relation_type",
            "/status",
            "/model_ids",
            "/scope",
            "/metric_labels",
            "/evaluation_run_ids",
            "/reporting_gaps",
            "/notes"
          ]
        }
      ],
      "id": "anthropic-scientific-figure-evaluation",
      "metric_labels": [
        "Accuracy delta"
      ],
      "model_ids": [
        "claude-opus-4-1",
        "claude-sonnet-4-5",
        "claude-opus-4-5"
      ],
      "notes": "Only the exact Opus 4.5 improvement annotation is normalized; plotted absolute values are not estimated.",
      "relation_type": "evaluation",
      "reporting_gaps": [
        "task count",
        "task data",
        "grader",
        "prompt",
        "tools",
        "repeats",
        "absolute accuracy"
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_reported",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-reported",
        "type": "unknown"
      },
      "status": "normalized",
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Exact relationship and delta are public while the underlying evaluation remains private.",
        "status": "verified"
      },
      "work_id": "anthropic-healthcare-life-sciences",
      "work_version_id": "anthropic-healthcare-life-sciences-2026-01-11"
    },
    {
      "benchmark_id": "spatialbench",
      "benchmark_version": "paper-v2",
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "anthropic-spatialbench-external-summary-evidence",
          "locator": {
            "note": "Caption says Source: LatchBio SpatialBench and 146 verifiable problems across five platforms and seven task categories.",
            "type": "figure",
            "value": "SpatialBench: Spatial biology analysis by LatchBio"
          },
          "source_id": "anthropic-healthcare-life-sciences",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/relation_type",
            "/status",
            "/model_ids",
            "/scope",
            "/metric_labels",
            "/evaluation_run_ids",
            "/reporting_gaps",
            "/notes"
          ]
        }
      ],
      "id": "anthropic-spatialbench-external-summary",
      "metric_labels": [
        "Accuracy"
      ],
      "model_ids": [
        "claude-opus-4-5",
        "claude-sonnet-4-5",
        "gpt-5-2",
        "gpt-5-1",
        "gemini-2-5-pro",
        "grok-4",
        "spatialbench-grok-4-1-unversioned"
      ],
      "notes": "The exact rounded chart values are explicitly attributed to LatchBio SpatialBench. This relation is a third-party result summary and is never counted as an Anthropic self-evaluation.",
      "relation_type": "external-result-summary",
      "reporting_gaps": [
        "Anthropic did not conduct or claim an independent rerun",
        "harness settings are inherited from the cited LatchBio source rather than reported as an Anthropic protocol"
      ],
      "scope": {
        "n": 146,
        "reporting_status": "reported",
        "selection": null,
        "selection_method": "all 146 paper-v2 problems described in the chart",
        "subset_id": null,
        "subset_kind": "not-applicable",
        "type": "full"
      },
      "status": "partial",
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Attribution and 146-problem scope are explicit on the official Anthropic page.",
        "status": "verified"
      },
      "work_id": "anthropic-healthcare-life-sciences",
      "work_version_id": "anthropic-healthcare-life-sciences-2026-01-11"
    },
    {
      "benchmark_id": "biosecbench-surveillance",
      "benchmark_version": "initial-release",
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [
        "biosecbench-8d53fd8-claude-code"
      ],
      "evidence": [
        {
          "accessed_date": "2026-08-11",
          "id": "biosecbench-8d53fd8-claude-code-use-evidence",
          "locator": {
            "note": "Full-scope Claude Code evaluation relation, models, metric, and linked run.",
            "type": "repository-path",
            "value": "METHODS.md and results/config_results.csv at commit 8d53fd8517cc74202eb18b618e8b39b4ffaf0c87"
          },
          "source_id": "biosecbench-surveillance-official-repository-resource",
          "source_type": "resource",
          "supports": [
            "/benchmark_version",
            "/relation_type",
            "/status",
            "/model_ids",
            "/scope",
            "/metric_labels",
            "/evaluation_run_ids",
            "/notes"
          ]
        }
      ],
      "id": "biosecbench-8d53fd8-claude-code-use",
      "metric_labels": [
        "endpoint pass rate"
      ],
      "model_ids": [
        "anthropic-opus-4-6",
        "anthropic-opus-4-7",
        "anthropic-opus-4-8",
        "anthropic-sonnet-4-6"
      ],
      "notes": "Official commit-pinned Claude Code evaluation result snapshot.",
      "relation_type": "evaluation",
      "reporting_gaps": [],
      "scope": {
        "n": 100,
        "reporting_status": "reported",
        "selection": null,
        "selection_method": "All 100 registered evaluations",
        "subset_id": null,
        "subset_kind": "not-applicable",
        "type": "full"
      },
      "status": "normalized",
      "verification": {
        "last_verified": "2026-08-11",
        "notes": "Normalized relationship is limited to independently verified high-confidence claims.",
        "status": "verified"
      },
      "work_id": "biosecbench-surveillance-repository-result-snapshot",
      "work_version_id": "biosecbench-surveillance-repository-result-snapshot-8d53fd8"
    },
    {
      "benchmark_id": "biosecbench-surveillance",
      "benchmark_version": "initial-release",
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [
        "biosecbench-8d53fd8-openai-codex"
      ],
      "evidence": [
        {
          "accessed_date": "2026-08-11",
          "id": "biosecbench-8d53fd8-openai-codex-use-evidence",
          "locator": {
            "note": "Full-scope OpenAI Codex evaluation relation, models, metric, and linked run.",
            "type": "repository-path",
            "value": "METHODS.md and results/config_results.csv at commit 8d53fd8517cc74202eb18b618e8b39b4ffaf0c87"
          },
          "source_id": "biosecbench-surveillance-official-repository-resource",
          "source_type": "resource",
          "supports": [
            "/benchmark_version",
            "/relation_type",
            "/status",
            "/model_ids",
            "/scope",
            "/metric_labels",
            "/evaluation_run_ids",
            "/notes"
          ]
        }
      ],
      "id": "biosecbench-8d53fd8-openai-codex-use",
      "metric_labels": [
        "endpoint pass rate"
      ],
      "model_ids": [
        "gpt-5-4",
        "gpt-5-5"
      ],
      "notes": "Official commit-pinned OpenAI Codex evaluation result snapshot.",
      "relation_type": "evaluation",
      "reporting_gaps": [],
      "scope": {
        "n": 100,
        "reporting_status": "reported",
        "selection": null,
        "selection_method": "All 100 registered evaluations",
        "subset_id": null,
        "subset_kind": "not-applicable",
        "type": "full"
      },
      "status": "normalized",
      "verification": {
        "last_verified": "2026-08-11",
        "notes": "Normalized relationship is limited to independently verified high-confidence claims.",
        "status": "verified"
      },
      "work_id": "biosecbench-surveillance-repository-result-snapshot",
      "work_version_id": "biosecbench-surveillance-repository-result-snapshot-8d53fd8"
    },
    {
      "benchmark_id": "biosecbench-surveillance",
      "benchmark_version": "initial-release",
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [
        "biosecbench-8d53fd8-pi"
      ],
      "evidence": [
        {
          "accessed_date": "2026-08-11",
          "id": "biosecbench-8d53fd8-pi-use-evidence",
          "locator": {
            "note": "Full-scope PI evaluation relation, models, metric, and linked run.",
            "type": "repository-path",
            "value": "METHODS.md and results/config_results.csv at commit 8d53fd8517cc74202eb18b618e8b39b4ffaf0c87"
          },
          "source_id": "biosecbench-surveillance-official-repository-resource",
          "source_type": "resource",
          "supports": [
            "/benchmark_version",
            "/relation_type",
            "/status",
            "/model_ids",
            "/scope",
            "/metric_labels",
            "/evaluation_run_ids",
            "/notes"
          ]
        }
      ],
      "id": "biosecbench-8d53fd8-pi-use",
      "metric_labels": [
        "endpoint pass rate"
      ],
      "model_ids": [
        "anthropic-opus-4-6",
        "anthropic-opus-4-7",
        "anthropic-opus-4-8",
        "anthropic-sonnet-4-6",
        "gemini-3-1-pro-preview",
        "gemini-3-5-flash",
        "gpt-5-4",
        "gpt-5-5",
        "grok-4-20-beta-0309-reasoning",
        "grok-4-3"
      ],
      "notes": "Official commit-pinned PI evaluation result snapshot.",
      "relation_type": "evaluation",
      "reporting_gaps": [],
      "scope": {
        "n": 100,
        "reporting_status": "reported",
        "selection": null,
        "selection_method": "All 100 registered evaluations",
        "subset_id": null,
        "subset_kind": "not-applicable",
        "type": "full"
      },
      "status": "normalized",
      "verification": {
        "last_verified": "2026-08-11",
        "notes": "Normalized relationship is limited to independently verified high-confidence claims.",
        "status": "verified"
      },
      "work_id": "biosecbench-surveillance-repository-result-snapshot",
      "work_version_id": "biosecbench-surveillance-repository-result-snapshot-8d53fd8"
    },
    {
      "benchmark_id": "biosecbench-surveillance",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-1-use-evidence-1",
          "locator": {
            "document_page": 1,
            "note": null,
            "printed_page": "1",
            "source_fragment_sha256": "d26f881c6f249284b8975be96ea8c8af9c7193804e78d3d7dfdffdbf4e2f6c7a",
            "type": "section",
            "value": "Abstract"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-1-use-evidence-2",
          "locator": {
            "document_page": 1,
            "note": null,
            "printed_page": "1",
            "source_fragment_sha256": "d8062a766e37ec1f470ba23e7973e72f44eff09d69f99acc298bd07bc84a5312",
            "type": "section",
            "value": "Abstract"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        }
      ],
      "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-1-use",
      "metric_labels": [],
      "model_ids": [],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "benchmark-creation",
      "reporting_gaps": [
        "Benchmark version is not reported.",
        "The exact public-subset size is not reported.",
        "The repository license is not reported."
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_applicable",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-applicable",
        "type": "unknown"
      },
      "status": "non-evaluation",
      "verification": {
        "last_verified": "2026-07-30",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
      "work_version_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-arxiv-v1"
    },
    {
      "benchmark_id": "biosecbench-surveillance",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-2-use-evidence-1",
          "locator": {
            "document_page": 4,
            "note": null,
            "printed_page": "4",
            "source_fragment_sha256": "3df4097dabfa21db61364876c9d9676d0a8d9cd560bdf319f1f19d4a6266c833",
            "type": "figure",
            "value": "Figure 2"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-2-use-evidence-2",
          "locator": {
            "document_page": 4,
            "note": null,
            "printed_page": "4",
            "source_fragment_sha256": "3df4097dabfa21db61364876c9d9676d0a8d9cd560bdf319f1f19d4a6266c833",
            "type": "figure",
            "value": "Figure 2"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-2-use-evidence-3",
          "locator": {
            "document_page": 7,
            "note": null,
            "printed_page": "7",
            "source_fragment_sha256": "6781fd905855d2b827bbb01794f899f45c14badbc40646c205e14b87035fc60d",
            "type": "section",
            "value": "Methods — Agent runs and execution"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-2-use-evidence-4",
          "locator": {
            "document_page": 7,
            "note": null,
            "printed_page": "7",
            "source_fragment_sha256": "6781fd905855d2b827bbb01794f899f45c14badbc40646c205e14b87035fc60d",
            "type": "section",
            "value": "Methods — Agent runs and execution"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-2-use-evidence-5",
          "locator": {
            "document_page": 7,
            "note": null,
            "printed_page": "7",
            "source_fragment_sha256": "84ca65ca353a7f5a918982c004730263c1d3f9a91d7ebc67e7d308cc53baac85",
            "type": "section",
            "value": "Methods — Outcome classification and aggregation"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-2-use-evidence-6",
          "locator": {
            "document_page": 4,
            "note": null,
            "printed_page": "4",
            "source_fragment_sha256": "be6cd2a9f5048ab8d328ef6ceb888b23dad867a3ac9f8a2832658597c2979512",
            "type": "figure",
            "value": "Figure 2"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-2-use-evidence-7",
          "locator": {
            "document_page": 4,
            "note": null,
            "printed_page": "4",
            "source_fragment_sha256": "f869c2b9df1fe2ab20b7ef9ce8fa476fc7c296f238a5a5c35374cc6e749a575c",
            "type": "figure",
            "value": "Figure 2"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-2-use-evidence-8",
          "locator": {
            "document_page": 4,
            "note": null,
            "printed_page": "4",
            "source_fragment_sha256": "50998ff79b4955ff17da5da352923aa2f5ede250f7abe808b70286e2a1389fe7",
            "type": "figure",
            "value": "Figure 2"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-2-use-evidence-9",
          "locator": {
            "document_page": 4,
            "note": null,
            "printed_page": "4",
            "source_fragment_sha256": "340ad3ab3e782bc1465305e6eb76bed70ba706fdf0f17b8dcdcc5a61262c2c16",
            "type": "figure",
            "value": "Figure 2"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-2-use-evidence-10",
          "locator": {
            "document_page": 4,
            "note": null,
            "printed_page": "4",
            "source_fragment_sha256": "74e23ca2dd4def58e0cc4b8497a7d08cbea17a22dc3872f434c1c92f051f811c",
            "type": "figure",
            "value": "Figure 2"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-2-use-evidence-11",
          "locator": {
            "document_page": 4,
            "note": null,
            "printed_page": "4",
            "source_fragment_sha256": "a58862cf05ed60f54528695ec23de4888c91754236c484e5342ba0ffedc99c80",
            "type": "figure",
            "value": "Figure 2"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-2-use-evidence-12",
          "locator": {
            "document_page": 4,
            "note": null,
            "printed_page": "4",
            "source_fragment_sha256": "67e1d458a3b5610411f02b6681ef5ab0421b95bb8f28456a03d16ffc6eece325",
            "type": "figure",
            "value": "Figure 2"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-2-use-evidence-13",
          "locator": {
            "document_page": 4,
            "note": null,
            "printed_page": "4",
            "source_fragment_sha256": "705eb255bf55583bc06857c2c5b7f81f76a01b66f38ef90f1351a30593592f3c",
            "type": "figure",
            "value": "Figure 2"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-2-use-evidence-14",
          "locator": {
            "document_page": 4,
            "note": null,
            "printed_page": "4",
            "source_fragment_sha256": "e45caaad286f4e46a4d61b6b875e9b2c52479b99288e6faf90745a4776b8752c",
            "type": "figure",
            "value": "Figure 2"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-2-use-evidence-15",
          "locator": {
            "document_page": 4,
            "note": null,
            "printed_page": "4",
            "source_fragment_sha256": "4e53e534af2903273d3a0551f9aff86e4d7be0b31655f2debdefdef30733345a",
            "type": "figure",
            "value": "Figure 2"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        }
      ],
      "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-2-use",
      "metric_labels": [
        "Endpoint pass rate"
      ],
      "model_ids": [
        "anthropic-opus-4-6",
        "anthropic-opus-4-7",
        "anthropic-opus-4-8",
        "anthropic-sonnet-4-6",
        "gemini-3-1-pro",
        "gemini-3-5-flash",
        "gpt-5-4",
        "gpt-5-5",
        "grok-4-3",
        "xai-grok-4-20-reasoning"
      ],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "evaluation",
      "reporting_gaps": [
        "Benchmark version is not reported.",
        "Exact deployment snapshots and model release dates are not reported.",
        "Prompt text, shots, token budget, and random seed are not reported.",
        "Gradable evaluation n is reported only for Opus 4.8 / PI.",
        "Exact confidence intervals are not numerically printed for seven PI results.",
        "benchmark version",
        "numeric result"
      ],
      "scope": {
        "n": 100,
        "reporting_status": "not_reported",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-reported",
        "type": "unknown"
      },
      "status": "partial",
      "verification": {
        "last_verified": "2026-07-30",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
      "work_version_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-arxiv-v1"
    },
    {
      "benchmark_id": "biosecbench-surveillance",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-3-use-evidence-1",
          "locator": {
            "document_page": 4,
            "note": null,
            "printed_page": "4",
            "source_fragment_sha256": "2f2afd31cb3ee815b0c8eb081afc6fce1d3234e0a0eb88b22c2f47b80d3e3cfc",
            "type": "figure",
            "value": "Figure 2"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-3-use-evidence-2",
          "locator": {
            "document_page": 4,
            "note": null,
            "printed_page": "4",
            "source_fragment_sha256": "3df4097dabfa21db61364876c9d9676d0a8d9cd560bdf319f1f19d4a6266c833",
            "type": "figure",
            "value": "Figure 2"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-3-use-evidence-3",
          "locator": {
            "document_page": 7,
            "note": null,
            "printed_page": "7",
            "source_fragment_sha256": "6781fd905855d2b827bbb01794f899f45c14badbc40646c205e14b87035fc60d",
            "type": "section",
            "value": "Methods — Agent runs and execution"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-3-use-evidence-4",
          "locator": {
            "document_page": 7,
            "note": null,
            "printed_page": "7",
            "source_fragment_sha256": "6781fd905855d2b827bbb01794f899f45c14badbc40646c205e14b87035fc60d",
            "type": "section",
            "value": "Methods — Agent runs and execution"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-3-use-evidence-5",
          "locator": {
            "document_page": 7,
            "note": null,
            "printed_page": "7",
            "source_fragment_sha256": "84ca65ca353a7f5a918982c004730263c1d3f9a91d7ebc67e7d308cc53baac85",
            "type": "section",
            "value": "Methods — Outcome classification and aggregation"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-3-use-evidence-6",
          "locator": {
            "document_page": 4,
            "note": null,
            "printed_page": "4",
            "source_fragment_sha256": "adf9b621ab7af3862e23627d6235c244211db29f72b11e583fff3f6527f0b1d8",
            "type": "figure",
            "value": "Figure 2"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-3-use-evidence-7",
          "locator": {
            "document_page": 4,
            "note": null,
            "printed_page": "4",
            "source_fragment_sha256": "953bca4161760b5e8d6d1a62d2c270de4838505cd2d2919d009baec744d99a3e",
            "type": "figure",
            "value": "Figure 2"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-3-use-evidence-8",
          "locator": {
            "document_page": 4,
            "note": null,
            "printed_page": "4",
            "source_fragment_sha256": "e908d26d21383bbd01c629ac3a72c84354702ddbe0e958beb2460a65922661ca",
            "type": "figure",
            "value": "Figure 2"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-3-use-evidence-9",
          "locator": {
            "document_page": 4,
            "note": null,
            "printed_page": "4",
            "source_fragment_sha256": "83ce27bf9e6eca048b65bf033c8b31890e2fb7ab1dc185b878ec8841f6e096a6",
            "type": "figure",
            "value": "Figure 2"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        }
      ],
      "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-3-use",
      "metric_labels": [
        "Endpoint pass rate"
      ],
      "model_ids": [
        "anthropic-opus-4-6",
        "anthropic-opus-4-7",
        "anthropic-opus-4-8",
        "anthropic-sonnet-4-6"
      ],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "evaluation",
      "reporting_gaps": [
        "Benchmark version is not reported.",
        "Exact deployment snapshots and model release dates are not reported.",
        "Prompt text, shots, token budget, and random seed are not reported.",
        "Per-configuration gradable n and numerical confidence intervals are not reported.",
        "benchmark version",
        "numeric result"
      ],
      "scope": {
        "n": 100,
        "reporting_status": "not_reported",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-reported",
        "type": "unknown"
      },
      "status": "partial",
      "verification": {
        "last_verified": "2026-07-30",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
      "work_version_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-arxiv-v1"
    },
    {
      "benchmark_id": "biosecbench-surveillance",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-4-use-evidence-1",
          "locator": {
            "document_page": 4,
            "note": null,
            "printed_page": "4",
            "source_fragment_sha256": "52ed484332b56ed40ec057169e25e50d0ad96a28f6a9891c3e121273f2b40646",
            "type": "figure",
            "value": "Figure 2"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-4-use-evidence-2",
          "locator": {
            "document_page": 4,
            "note": null,
            "printed_page": "4",
            "source_fragment_sha256": "3df4097dabfa21db61364876c9d9676d0a8d9cd560bdf319f1f19d4a6266c833",
            "type": "figure",
            "value": "Figure 2"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-4-use-evidence-3",
          "locator": {
            "document_page": 7,
            "note": null,
            "printed_page": "7",
            "source_fragment_sha256": "6781fd905855d2b827bbb01794f899f45c14badbc40646c205e14b87035fc60d",
            "type": "section",
            "value": "Methods — Agent runs and execution"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-4-use-evidence-4",
          "locator": {
            "document_page": 7,
            "note": null,
            "printed_page": "7",
            "source_fragment_sha256": "6781fd905855d2b827bbb01794f899f45c14badbc40646c205e14b87035fc60d",
            "type": "section",
            "value": "Methods — Agent runs and execution"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-4-use-evidence-5",
          "locator": {
            "document_page": 7,
            "note": null,
            "printed_page": "7",
            "source_fragment_sha256": "84ca65ca353a7f5a918982c004730263c1d3f9a91d7ebc67e7d308cc53baac85",
            "type": "section",
            "value": "Methods — Outcome classification and aggregation"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-4-use-evidence-6",
          "locator": {
            "document_page": 4,
            "note": null,
            "printed_page": "4",
            "source_fragment_sha256": "98c41fa22399a400cd3522f85519e25f8770e82a01f7c174f6444a0526f9dd95",
            "type": "figure",
            "value": "Figure 2"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-4-use-evidence-7",
          "locator": {
            "document_page": 4,
            "note": null,
            "printed_page": "4",
            "source_fragment_sha256": "98940101fb996a1bbd5288eb6a3ab6d9c3ed53fe2ca084767297f66e7e58fc54",
            "type": "figure",
            "value": "Figure 2"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        }
      ],
      "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-4-use",
      "metric_labels": [
        "Endpoint pass rate"
      ],
      "model_ids": [
        "gpt-5-4",
        "gpt-5-5"
      ],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "evaluation",
      "reporting_gaps": [
        "Benchmark version is not reported.",
        "Exact deployment snapshots and model release dates are not reported.",
        "Prompt text, shots, token budget, and random seed are not reported.",
        "Gradable evaluation n is not reported for either Codex configuration.",
        "GPT-5.4 / Codex has no numerically printed confidence interval.",
        "benchmark version",
        "numeric result"
      ],
      "scope": {
        "n": 100,
        "reporting_status": "not_reported",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-reported",
        "type": "unknown"
      },
      "status": "partial",
      "verification": {
        "last_verified": "2026-07-30",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
      "work_version_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-arxiv-v1"
    },
    {
      "benchmark_id": "comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-08-06",
          "id": "comprehensive-benchmark-of-differential-transcript-usa-comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-1-use-evidence-1",
          "locator": {
            "document_page": 2,
            "note": null,
            "printed_page": "1",
            "source_fragment_sha256": "23af8973357fb73f9a51af2ec8bd60de2b2e436353b92b422e8f95a29be8d221",
            "type": "section",
            "value": "Abstract"
          },
          "source_id": "comprehensive-benchmark-of-differential-transcript-usa",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-08-06",
          "id": "comprehensive-benchmark-of-differential-transcript-usa-comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-1-use-evidence-2",
          "locator": {
            "document_page": 2,
            "note": null,
            "printed_page": "1",
            "source_fragment_sha256": "e35160733c655a0caa8e257f1592e50e98d8594e3c2bd3fca558e7350684c6c0",
            "type": "page",
            "value": "Article title"
          },
          "source_id": "comprehensive-benchmark-of-differential-transcript-usa",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        }
      ],
      "id": "comprehensive-benchmark-of-differential-transcript-usa-comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-1-use",
      "metric_labels": [],
      "model_ids": [],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "benchmark-creation",
      "reporting_gaps": [
        "No single finite primary item inventory is reported for the reusable benchmark matrix.",
        "No benchmark version or separate release date is stated.",
        "The paper does not explicitly assign a Registry access level."
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_applicable",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-applicable",
        "type": "unknown"
      },
      "status": "non-evaluation",
      "verification": {
        "last_verified": "2026-08-06",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "comprehensive-benchmark-of-differential-transcript-usa",
      "work_version_id": "comprehensive-benchmark-of-differential-transcript-usa-final-published-version"
    },
    {
      "benchmark_id": "comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-08-06",
          "id": "comprehensive-benchmark-of-differential-transcript-usa-comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-2-use-evidence-1",
          "locator": {
            "document_page": 5,
            "note": null,
            "printed_page": "4",
            "source_fragment_sha256": "7e4af4cac21e4d3ce17e539273434f242da54e02beaad95d28720f5213f9a187",
            "type": "section",
            "value": "Overall performance of DTU detection"
          },
          "source_id": "comprehensive-benchmark-of-differential-transcript-usa",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-08-06",
          "id": "comprehensive-benchmark-of-differential-transcript-usa-comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-2-use-evidence-2",
          "locator": {
            "document_page": 2,
            "note": null,
            "printed_page": "1",
            "source_fragment_sha256": "e35160733c655a0caa8e257f1592e50e98d8594e3c2bd3fca558e7350684c6c0",
            "type": "page",
            "value": "Article title"
          },
          "source_id": "comprehensive-benchmark-of-differential-transcript-usa",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        }
      ],
      "id": "comprehensive-benchmark-of-differential-transcript-usa-comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-2-use",
      "metric_labels": [],
      "model_ids": [],
      "notes": "Owner-reviewed conservative publication: the creator evaluation is retained only as a partial relationship; conflicted settings and outcomes are omitted pending manual reconciliation.",
      "relation_type": "evaluation",
      "reporting_gaps": [
        "The parameter matrix does not provide one uniform realized evaluation n.",
        "Random seeds and versions of most evaluated DTU methods are not reported.",
        "Unlabeled graphical scores were not estimated.",
        "benchmark version",
        "realized n/scope",
        "metric",
        "numeric result",
        "prompt and tools",
        "grader and repeats",
        "exact model"
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_reported",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-reported",
        "type": "unknown"
      },
      "status": "partial",
      "verification": {
        "last_verified": "2026-08-06",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "comprehensive-benchmark-of-differential-transcript-usa",
      "work_version_id": "comprehensive-benchmark-of-differential-transcript-usa-final-published-version"
    },
    {
      "benchmark_id": "comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-08-06",
          "id": "comprehensive-benchmark-of-differential-transcript-usa-comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-3-use-evidence-1",
          "locator": {
            "document_page": 9,
            "note": null,
            "printed_page": "8",
            "source_fragment_sha256": "9ea35ab9d7a89f705a89e65c4c95b2651155b7c2333affc8522c5bc207de1955",
            "type": "section",
            "value": "Performance in single-cell data"
          },
          "source_id": "comprehensive-benchmark-of-differential-transcript-usa",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-08-06",
          "id": "comprehensive-benchmark-of-differential-transcript-usa-comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-3-use-evidence-2",
          "locator": {
            "document_page": 2,
            "note": null,
            "printed_page": "1",
            "source_fragment_sha256": "e35160733c655a0caa8e257f1592e50e98d8594e3c2bd3fca558e7350684c6c0",
            "type": "page",
            "value": "Article title"
          },
          "source_id": "comprehensive-benchmark-of-differential-transcript-usa",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        }
      ],
      "id": "comprehensive-benchmark-of-differential-transcript-usa-comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-3-use",
      "metric_labels": [],
      "model_ids": [],
      "notes": "Owner-reviewed conservative publication: the creator evaluation is retained only as a partial relationship; conflicted settings and outcomes are omitted pending manual reconciliation.",
      "relation_type": "evaluation",
      "reporting_gaps": [
        "Simulation repeat count and random seed are not reported.",
        "The varying balanced and unbalanced dataset sizes prevent one uniform realized n.",
        "benchmark version",
        "realized n/scope",
        "metric",
        "numeric result",
        "prompt and tools",
        "grader and repeats",
        "exact model"
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_reported",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-reported",
        "type": "unknown"
      },
      "status": "partial",
      "verification": {
        "last_verified": "2026-08-06",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "comprehensive-benchmark-of-differential-transcript-usa",
      "work_version_id": "comprehensive-benchmark-of-differential-transcript-usa-final-published-version"
    },
    {
      "benchmark_id": "comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-08-06",
          "id": "comprehensive-benchmark-of-differential-transcript-usa-comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-4-use-evidence-1",
          "locator": {
            "document_page": 9,
            "note": null,
            "printed_page": "8",
            "source_fragment_sha256": "669bbfb6872beba51b78a44bff53f6caaf04ba80f0434f795334ed91cba5f490",
            "type": "section",
            "value": "Performance in single-cell data"
          },
          "source_id": "comprehensive-benchmark-of-differential-transcript-usa",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-08-06",
          "id": "comprehensive-benchmark-of-differential-transcript-usa-comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-4-use-evidence-2",
          "locator": {
            "document_page": 2,
            "note": null,
            "printed_page": "1",
            "source_fragment_sha256": "e35160733c655a0caa8e257f1592e50e98d8594e3c2bd3fca558e7350684c6c0",
            "type": "page",
            "value": "Article title"
          },
          "source_id": "comprehensive-benchmark-of-differential-transcript-usa",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        }
      ],
      "id": "comprehensive-benchmark-of-differential-transcript-usa-comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-4-use",
      "metric_labels": [],
      "model_ids": [],
      "notes": "Owner-reviewed conservative publication: the creator evaluation is retained only as a partial relationship; conflicted settings and outcomes are omitted pending manual reconciliation.",
      "relation_type": "evaluation",
      "reporting_gaps": [
        "Simulation repeat count and random seed are not reported.",
        "The unbalanced meta-cell evaluation has no single reported realized cell count.",
        "benchmark version",
        "realized n/scope",
        "metric",
        "numeric result",
        "prompt and tools",
        "grader and repeats",
        "exact model"
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_reported",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-reported",
        "type": "unknown"
      },
      "status": "partial",
      "verification": {
        "last_verified": "2026-08-06",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "comprehensive-benchmark-of-differential-transcript-usa",
      "work_version_id": "comprehensive-benchmark-of-differential-transcript-usa-final-published-version"
    },
    {
      "benchmark_id": "comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-08-06",
          "id": "comprehensive-benchmark-of-differential-transcript-usa-comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-5-use-evidence-1",
          "locator": {
            "document_page": 7,
            "note": null,
            "printed_page": "6",
            "source_fragment_sha256": "3dd15740dc6d7771880bdafa802a058cfaf0c854cdd919a704a4fd8637b88b97",
            "type": "section",
            "value": "Benchmarking with a real transcriptome dataset"
          },
          "source_id": "comprehensive-benchmark-of-differential-transcript-usa",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-08-06",
          "id": "comprehensive-benchmark-of-differential-transcript-usa-comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-5-use-evidence-2",
          "locator": {
            "document_page": 2,
            "note": null,
            "printed_page": "1",
            "source_fragment_sha256": "e35160733c655a0caa8e257f1592e50e98d8594e3c2bd3fca558e7350684c6c0",
            "type": "page",
            "value": "Article title"
          },
          "source_id": "comprehensive-benchmark-of-differential-transcript-usa",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        }
      ],
      "id": "comprehensive-benchmark-of-differential-transcript-usa-comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-5-use",
      "metric_labels": [],
      "model_ids": [],
      "notes": "Owner-reviewed conservative publication: the creator evaluation is retained only as a partial relationship; conflicted settings and outcomes are omitted pending manual reconciliation.",
      "relation_type": "evaluation",
      "reporting_gaps": [
        "The realized sample count after removing 10 matching samples is not explicitly reported.",
        "The real dataset has no simulated ground truth.",
        "F1 dots without printed numeric labels were not estimated.",
        "benchmark version",
        "realized n/scope",
        "metric",
        "numeric result",
        "prompt and tools",
        "grader and repeats",
        "exact model"
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_reported",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-reported",
        "type": "unknown"
      },
      "status": "partial",
      "verification": {
        "last_verified": "2026-08-06",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "comprehensive-benchmark-of-differential-transcript-usa",
      "work_version_id": "comprehensive-benchmark-of-differential-transcript-usa-final-published-version"
    },
    {
      "benchmark_id": "comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-08-06",
          "id": "comprehensive-benchmark-of-differential-transcript-usa-comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-6-use-evidence-1",
          "locator": {
            "document_page": 9,
            "note": null,
            "printed_page": "8",
            "source_fragment_sha256": "899bbaa6c11ea33537beefe35116968de42a3c312c19a9bd07fa399a381b0d50",
            "type": "section",
            "value": "Time series IS analysis"
          },
          "source_id": "comprehensive-benchmark-of-differential-transcript-usa",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-08-06",
          "id": "comprehensive-benchmark-of-differential-transcript-usa-comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-6-use-evidence-2",
          "locator": {
            "document_page": 2,
            "note": null,
            "printed_page": "1",
            "source_fragment_sha256": "e35160733c655a0caa8e257f1592e50e98d8594e3c2bd3fca558e7350684c6c0",
            "type": "page",
            "value": "Article title"
          },
          "source_id": "comprehensive-benchmark-of-differential-transcript-usa",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        }
      ],
      "id": "comprehensive-benchmark-of-differential-transcript-usa-comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-6-use",
      "metric_labels": [],
      "model_ids": [],
      "notes": "Owner-reviewed conservative publication: the creator evaluation is retained only as a partial relationship; conflicted settings and outcomes are omitted pending manual reconciliation.",
      "relation_type": "evaluation",
      "reporting_gaps": [
        "The body and Figure 7 caption disagree about the evaluated method list.",
        "No ground-truth performance metric is reported for the real time-series dataset.",
        "Conflicted scope-n claim omitted after independent verification; the evaluation relationship is published conservatively.",
        "Conflicted result claim omitted after independent verification; the evaluation relationship is published conservatively.",
        "benchmark version",
        "realized n/scope",
        "metric",
        "numeric result",
        "prompt and tools",
        "grader and repeats",
        "exact model"
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_reported",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-reported",
        "type": "unknown"
      },
      "status": "partial",
      "verification": {
        "last_verified": "2026-08-06",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "comprehensive-benchmark-of-differential-transcript-usa",
      "work_version_id": "comprehensive-benchmark-of-differential-transcript-usa-final-published-version"
    },
    {
      "benchmark_id": "crafted-experiments",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-31",
          "id": "crafted-experiments-to-evaluate-feature-selection-meth-crafted-experiments-1-use-evidence-1",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": "lqaf023",
            "source_fragment_sha256": "704e0de5dc664413c19d895d12b82fd9c86283ba443eea2782d438ded3f09ebe",
            "type": "section",
            "value": "Introduction"
          },
          "source_id": "crafted-experiments-to-evaluate-feature-selection-meth",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "crafted-experiments-to-evaluate-feature-selection-meth-crafted-experiments-1-use-evidence-2",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": "lqaf023",
            "source_fragment_sha256": "047411fa0329617e06c84ef40f453a55b21e288d3ebbca40a5cf352261a3083b",
            "type": "section",
            "value": "Introduction"
          },
          "source_id": "crafted-experiments-to-evaluate-feature-selection-meth",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        }
      ],
      "id": "crafted-experiments-to-evaluate-feature-selection-meth-crafted-experiments-1-use",
      "metric_labels": [],
      "model_ids": [],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "benchmark-creation",
      "reporting_gaps": [
        "Benchmark version is not reported.",
        "An explicit benchmark release date is not reported.",
        "The released crafted-dataset artifact's license is not reported."
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_applicable",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-applicable",
        "type": "unknown"
      },
      "status": "non-evaluation",
      "verification": {
        "last_verified": "2026-07-31",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "crafted-experiments-to-evaluate-feature-selection-meth",
      "work_version_id": "crafted-experiments-to-evaluate-feature-selection-meth-2025-03-19"
    },
    {
      "benchmark_id": "crafted-experiments",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-31",
          "id": "crafted-experiments-to-evaluate-feature-selection-meth-crafted-experiments-2-use-evidence-1",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": "lqaf023",
            "source_fragment_sha256": "5cf2700309cfac598c253a9540a564c1ca92dfc68a4b3cc4c53a12d3d7f84efd",
            "type": "section",
            "value": "Results — Crafted experiment applications"
          },
          "source_id": "crafted-experiments-to-evaluate-feature-selection-meth",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "crafted-experiments-to-evaluate-feature-selection-meth-crafted-experiments-2-use-evidence-2",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": "lqaf023",
            "source_fragment_sha256": "5a1cb25d9939c19ab1450d7d67f9513ea92202b970e78ddab257449ecd67600a",
            "type": "section",
            "value": "Results — Crafted experiment applications"
          },
          "source_id": "crafted-experiments-to-evaluate-feature-selection-meth",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "crafted-experiments-to-evaluate-feature-selection-meth-crafted-experiments-2-use-evidence-3",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": "lqaf023",
            "source_fragment_sha256": "85d844abdf398d64fccb4c6d79b56945059d5e7db7afb93eb1e2967e47c85b7a",
            "type": "section",
            "value": "Results — Crafted experiment applications"
          },
          "source_id": "crafted-experiments-to-evaluate-feature-selection-meth",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "crafted-experiments-to-evaluate-feature-selection-meth-crafted-experiments-2-use-evidence-4",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": "lqaf023",
            "source_fragment_sha256": "85d844abdf398d64fccb4c6d79b56945059d5e7db7afb93eb1e2967e47c85b7a",
            "type": "section",
            "value": "Results — Crafted experiment applications"
          },
          "source_id": "crafted-experiments-to-evaluate-feature-selection-meth",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "crafted-experiments-to-evaluate-feature-selection-meth-crafted-experiments-2-use-evidence-5",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": "lqaf023",
            "source_fragment_sha256": "85d844abdf398d64fccb4c6d79b56945059d5e7db7afb93eb1e2967e47c85b7a",
            "type": "section",
            "value": "Results — Crafted experiment applications"
          },
          "source_id": "crafted-experiments-to-evaluate-feature-selection-meth",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "crafted-experiments-to-evaluate-feature-selection-meth-crafted-experiments-2-use-evidence-6",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": "lqaf023",
            "source_fragment_sha256": "482341b340f4918d3fd6da929dcb2d2c7c31434aba9b270e82becd71814dbb71",
            "type": "section",
            "value": "Materials and methods — Benchmarking of feature selection methods"
          },
          "source_id": "crafted-experiments-to-evaluate-feature-selection-meth",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "crafted-experiments-to-evaluate-feature-selection-meth-crafted-experiments-2-use-evidence-7",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": "lqaf023",
            "source_fragment_sha256": "ec12e31d16fcd8bc8fad30956330893b093fac7bf421c2f3830bf55d16f56564",
            "type": "section",
            "value": "Results — Crafted experiment applications; Reference 16"
          },
          "source_id": "crafted-experiments-to-evaluate-feature-selection-meth",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "crafted-experiments-to-evaluate-feature-selection-meth-crafted-experiments-2-use-evidence-8",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": "lqaf023",
            "source_fragment_sha256": "c8c19edb9405a506d936563cd3381b5d590016e38bf00674715569ec288535ee",
            "type": "section",
            "value": "Results — Crafted experiment applications; Reference 15"
          },
          "source_id": "crafted-experiments-to-evaluate-feature-selection-meth",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "crafted-experiments-to-evaluate-feature-selection-meth-crafted-experiments-2-use-evidence-9",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": "lqaf023",
            "source_fragment_sha256": "eaf3a2b1882b0f1df4972d1fdb3a31660e860fe25532a36367d23a09118a84f0",
            "type": "section",
            "value": "Results — Crafted experiment applications; Reference 18"
          },
          "source_id": "crafted-experiments-to-evaluate-feature-selection-meth",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "crafted-experiments-to-evaluate-feature-selection-meth-crafted-experiments-2-use-evidence-10",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": "lqaf023",
            "source_fragment_sha256": "6235a0fb4e237af39042cd6a938cdd9a380fd21ddca57af4d9894ac9f215230e",
            "type": "section",
            "value": "Results — Crafted experiment applications"
          },
          "source_id": "crafted-experiments-to-evaluate-feature-selection-meth",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "crafted-experiments-to-evaluate-feature-selection-meth-crafted-experiments-2-use-evidence-11",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": "lqaf023",
            "source_fragment_sha256": "d0e17d2a7ea5d18ba4a140dd1d2020079731d726fabf9294db94ed0ae1a72ebb",
            "type": "other",
            "value": "Article author list; Results — Crafted experiment applications"
          },
          "source_id": "crafted-experiments-to-evaluate-feature-selection-meth",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "crafted-experiments-to-evaluate-feature-selection-meth-crafted-experiments-2-use-evidence-12",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": "lqaf023",
            "source_fragment_sha256": "d8d00cec3e01531fdf0e5a821c982f5557b8701a1f60af9bc565e641a1cac41e",
            "type": "other",
            "value": "Article author list; Results — Crafted experiment applications"
          },
          "source_id": "crafted-experiments-to-evaluate-feature-selection-meth",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "crafted-experiments-to-evaluate-feature-selection-meth-crafted-experiments-2-use-evidence-13",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": "lqaf023",
            "source_fragment_sha256": "f20ecfefa486d69e86934df200115dfe67abe29ba1cf34b67b2b25142f1e7128",
            "type": "other",
            "value": "Article author list; Results — Crafted experiment applications"
          },
          "source_id": "crafted-experiments-to-evaluate-feature-selection-meth",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "crafted-experiments-to-evaluate-feature-selection-meth-crafted-experiments-2-use-evidence-14",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": "lqaf023",
            "source_fragment_sha256": "01114f672b5e804197ecb261efb5e7f68fa48380a9ada0c52788e9c78dac9314",
            "type": "section",
            "value": "Materials and methods — Benchmarking of feature selection methods"
          },
          "source_id": "crafted-experiments-to-evaluate-feature-selection-meth",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "crafted-experiments-to-evaluate-feature-selection-meth-crafted-experiments-2-use-evidence-15",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": "lqaf023",
            "source_fragment_sha256": "45846970b35ce447aa579c5536527741193f5c044d1bf139326d9de3798ac353",
            "type": "section",
            "value": "Materials and methods — Benchmarking of feature selection methods"
          },
          "source_id": "crafted-experiments-to-evaluate-feature-selection-meth",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "crafted-experiments-to-evaluate-feature-selection-meth-crafted-experiments-2-use-evidence-16",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": "lqaf023",
            "source_fragment_sha256": "2254320545defb334e1681505a6bb482e8de329ef0078d6191a4baae8f610c97",
            "type": "section",
            "value": "Results — Crafted experiment applications"
          },
          "source_id": "crafted-experiments-to-evaluate-feature-selection-meth",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "crafted-experiments-to-evaluate-feature-selection-meth-crafted-experiments-2-use-evidence-17",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": "lqaf023",
            "source_fragment_sha256": "d65a4c2f88ca04e2de1729a006885cfaf6346a8567bd1ada1c1dbcd37b7f8c0b",
            "type": "section",
            "value": "Results — Crafted experiment applications"
          },
          "source_id": "crafted-experiments-to-evaluate-feature-selection-meth",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "crafted-experiments-to-evaluate-feature-selection-meth-crafted-experiments-2-use-evidence-18",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": "lqaf023",
            "source_fragment_sha256": "6ce0c15331b659e4ee917e57a8bd2515365dfe11fb3c961d0456ad591a48a68e",
            "type": "section",
            "value": "Results — Crafted experiment applications"
          },
          "source_id": "crafted-experiments-to-evaluate-feature-selection-meth",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "crafted-experiments-to-evaluate-feature-selection-meth-crafted-experiments-2-use-evidence-19",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": "lqaf023",
            "source_fragment_sha256": "6ce0c15331b659e4ee917e57a8bd2515365dfe11fb3c961d0456ad591a48a68e",
            "type": "section",
            "value": "Results — Crafted experiment applications"
          },
          "source_id": "crafted-experiments-to-evaluate-feature-selection-meth",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        }
      ],
      "id": "crafted-experiments-to-evaluate-feature-selection-meth-crafted-experiments-2-use",
      "metric_labels": [
        "MultiK low-resolution rank",
        "DiProPerm Z scores",
        "average rank",
        "proportion of crafted genes selected",
        "number of crafted genes selected",
        "number of crafted genes not selected"
      ],
      "model_ids": [
        "kim-zhou-and-chen-hippo",
        "liu-et-al-pp-anb",
        "liu-et-al-qq-anb",
        "liu-et-al-wdist-med",
        "satija-et-al-seurat-disp",
        "stuart-et-al-seurat-vst",
        "townes-et-al-deviancefs"
      ],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "evaluation",
      "reporting_gaps": [
        "Benchmark version is not reported.",
        "Exact software versions and random seeds are not reported.",
        "Figure 4 bar values are unlabeled, so no numeric model results were extracted.",
        "Conflicted model claim omitted after independent verification; the evaluation relationship is published conservatively.",
        "Conflicted repeats claim omitted after independent verification; the evaluation relationship is published conservatively.",
        "benchmark version",
        "numeric result"
      ],
      "scope": {
        "n": 24,
        "reporting_status": "not_reported",
        "selection": null,
        "selection_method": "Guided tutorials and default parameters; top 2000 ranked features per method, except HIPPO used its zero-proportion-test cutoff",
        "subset_id": null,
        "subset_kind": "not-reported",
        "type": "unknown"
      },
      "status": "partial",
      "verification": {
        "last_verified": "2026-07-31",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "crafted-experiments-to-evaluate-feature-selection-meth",
      "work_version_id": "crafted-experiments-to-evaluate-feature-selection-meth-2025-03-19"
    },
    {
      "benchmark_id": "moleculenet",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-08-02",
          "id": "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-1-use-evidence-1",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "609aebc5dbc5994274443e5d05466fc077457deab049f16c55eade7f0a0553a5",
            "type": "section",
            "value": "Results and discussion > Experimental setup"
          },
          "source_id": "enhancing-molecular-property-prediction-with-auxiliary",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-1-use-evidence-2",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "d980d38df62519335682b1964134b0b656d479c9f7c490e8736c48b95492f4ea",
            "type": "section",
            "value": "Results and discussion > Experimental setup"
          },
          "source_id": "enhancing-molecular-property-prediction-with-auxiliary",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-1-use-evidence-3",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "d980d38df62519335682b1964134b0b656d479c9f7c490e8736c48b95492f4ea",
            "type": "section",
            "value": "Results and discussion > Experimental setup"
          },
          "source_id": "enhancing-molecular-property-prediction-with-auxiliary",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-1-use-evidence-4",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "d980d38df62519335682b1964134b0b656d479c9f7c490e8736c48b95492f4ea",
            "type": "section",
            "value": "Results and discussion > Experimental setup"
          },
          "source_id": "enhancing-molecular-property-prediction-with-auxiliary",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-1-use-evidence-5",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "ce83e1ac92fd5c60627da35cf459bdb4f02db3e2f03d650e12c87c190ad50a45",
            "type": "table",
            "value": "Table 4, dataset columns"
          },
          "source_id": "enhancing-molecular-property-prediction-with-auxiliary",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-1-use-evidence-6",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "8e19f2f648510df76c5f9b958ebd2e94207d5a8eb90aa81c8b47987fb98a28ed",
            "type": "section",
            "value": "Results and discussion > Reproducibility and implementation details"
          },
          "source_id": "enhancing-molecular-property-prediction-with-auxiliary",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-1-use-evidence-7",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "22aea0bb5f9afc4285c7787ad872b14ce52ae91697d1ec2a436feebb94ee4ea4",
            "type": "other",
            "value": "Experimental setup; reference 17"
          },
          "source_id": "enhancing-molecular-property-prediction-with-auxiliary",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-1-use-evidence-8",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "a03c4558af1a1f6d24f6feec8d955744b46edecf210c5ef1d0f146cbb1250bfc",
            "type": "other",
            "value": "Experimental setup; reference 17"
          },
          "source_id": "enhancing-molecular-property-prediction-with-auxiliary",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        }
      ],
      "id": "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-1-use",
      "metric_labels": [],
      "model_ids": [
        "hu-et-al-supervised-contextpred-sup-cp",
        "hu-et-al-supervised-sup"
      ],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "fine-tuning",
      "reporting_gaps": [
        "MoleculeNet version is not reported.",
        "Table 4 prints the Tox21 molecule count ambiguously as 7.831."
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_applicable",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-applicable",
        "type": "unknown"
      },
      "status": "non-evaluation",
      "verification": {
        "last_verified": "2026-08-02",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "enhancing-molecular-property-prediction-with-auxiliary",
      "work_version_id": "enhancing-molecular-property-prediction-with-auxiliary-pmc-version-1"
    },
    {
      "benchmark_id": "moleculenet",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-08-02",
          "id": "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-2-use-evidence-1",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "b1be8dbf9d0d717ab2ffb681db68460a310361facf4e144b0b3178b6911cf043",
            "type": "table",
            "value": "Table 1"
          },
          "source_id": "enhancing-molecular-property-prediction-with-auxiliary",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-2-use-evidence-2",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "d980d38df62519335682b1964134b0b656d479c9f7c490e8736c48b95492f4ea",
            "type": "section",
            "value": "Results and discussion > Experimental setup"
          },
          "source_id": "enhancing-molecular-property-prediction-with-auxiliary",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-2-use-evidence-3",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "d980d38df62519335682b1964134b0b656d479c9f7c490e8736c48b95492f4ea",
            "type": "section",
            "value": "Results and discussion > Experimental setup"
          },
          "source_id": "enhancing-molecular-property-prediction-with-auxiliary",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-2-use-evidence-4",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "d980d38df62519335682b1964134b0b656d479c9f7c490e8736c48b95492f4ea",
            "type": "section",
            "value": "Results and discussion > Experimental setup"
          },
          "source_id": "enhancing-molecular-property-prediction-with-auxiliary",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-2-use-evidence-5",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "1be2ace855d6b72411fc8616ecadc846eaa3a9bd399ec7f33e242b91fe899d13",
            "type": "table",
            "value": "Table 1, dataset columns"
          },
          "source_id": "enhancing-molecular-property-prediction-with-auxiliary",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-2-use-evidence-6",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "d233a88f7cdc22f8e834a4355dbd96d3e023033a0477dd7dcc0dc3b2cc938f4e",
            "type": "table",
            "value": "Table 1 footnote"
          },
          "source_id": "enhancing-molecular-property-prediction-with-auxiliary",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-2-use-evidence-7",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "22aea0bb5f9afc4285c7787ad872b14ce52ae91697d1ec2a436feebb94ee4ea4",
            "type": "other",
            "value": "Experimental setup; reference 17"
          },
          "source_id": "enhancing-molecular-property-prediction-with-auxiliary",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-2-use-evidence-8",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "cce5b72bf4a47c6b1fe42331469e628c80f67c5cbb47bdc9ccc1affe780c981b",
            "type": "table",
            "value": "Table 1 caption and footnote"
          },
          "source_id": "enhancing-molecular-property-prediction-with-auxiliary",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        }
      ],
      "id": "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-2-use",
      "metric_labels": [
        "Test ROC-AUC (Sup-CP; AM,CP,EP,IG,MP)"
      ],
      "model_ids": [
        "hu-et-al-supervised-contextpred-sup-cp"
      ],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "evaluation",
      "reporting_gaps": [
        "MoleculeNet version is not reported.",
        "Realized test-set sample sizes are not reported.",
        "Exact checkpoint revision and release date are not reported.",
        "Scaffold splitting is the within-dataset split protocol, not the method used to select the eight benchmark datasets.",
        "Table rows vary by adaptation strategy; those strategies are not normalized here as standalone model identities.",
        "Conflicted result claim omitted after independent verification; the evaluation relationship is published conservatively.",
        "benchmark version",
        "numeric result"
      ],
      "scope": {
        "n": 8,
        "reporting_status": "reported",
        "selection": "filtered",
        "selection_method": "Eight MoleculeNet classification datasets: BBBP, Tox21, ToxCast, SIDER, ClinTox, MUV, HIV, and BACE.",
        "subset_id": null,
        "subset_kind": "paper-specific",
        "type": "subset"
      },
      "status": "partial",
      "verification": {
        "last_verified": "2026-08-02",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "enhancing-molecular-property-prediction-with-auxiliary",
      "work_version_id": "enhancing-molecular-property-prediction-with-auxiliary-pmc-version-1"
    },
    {
      "benchmark_id": "moleculenet",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-08-02",
          "id": "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-3-use-evidence-1",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "0c35e9a59735d17c5e67b413c2c4cad6da2f4768bd9779ed57652c0a244b49e2",
            "type": "table",
            "value": "Table 2"
          },
          "source_id": "enhancing-molecular-property-prediction-with-auxiliary",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-3-use-evidence-2",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "d980d38df62519335682b1964134b0b656d479c9f7c490e8736c48b95492f4ea",
            "type": "section",
            "value": "Results and discussion > Experimental setup"
          },
          "source_id": "enhancing-molecular-property-prediction-with-auxiliary",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-3-use-evidence-3",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "d980d38df62519335682b1964134b0b656d479c9f7c490e8736c48b95492f4ea",
            "type": "section",
            "value": "Results and discussion > Experimental setup"
          },
          "source_id": "enhancing-molecular-property-prediction-with-auxiliary",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-3-use-evidence-4",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "d980d38df62519335682b1964134b0b656d479c9f7c490e8736c48b95492f4ea",
            "type": "section",
            "value": "Results and discussion > Experimental setup"
          },
          "source_id": "enhancing-molecular-property-prediction-with-auxiliary",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-3-use-evidence-5",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "1be2ace855d6b72411fc8616ecadc846eaa3a9bd399ec7f33e242b91fe899d13",
            "type": "table",
            "value": "Table 2, dataset columns"
          },
          "source_id": "enhancing-molecular-property-prediction-with-auxiliary",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-3-use-evidence-6",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "8e19f2f648510df76c5f9b958ebd2e94207d5a8eb90aa81c8b47987fb98a28ed",
            "type": "section",
            "value": "Results and discussion > Reproducibility and implementation details"
          },
          "source_id": "enhancing-molecular-property-prediction-with-auxiliary",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-3-use-evidence-7",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "22aea0bb5f9afc4285c7787ad872b14ce52ae91697d1ec2a436feebb94ee4ea4",
            "type": "other",
            "value": "Experimental setup; reference 17"
          },
          "source_id": "enhancing-molecular-property-prediction-with-auxiliary",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-3-use-evidence-8",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "b1c9789e729a3d36992012efd84095cb0e84ec97225fb3eee7c83095d8dc1b6e",
            "type": "table",
            "value": "Table 2 caption and footnote"
          },
          "source_id": "enhancing-molecular-property-prediction-with-auxiliary",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        }
      ],
      "id": "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-3-use",
      "metric_labels": [
        "Test ROC-AUC (Sup-CP; AM,IG,MP)"
      ],
      "model_ids": [
        "hu-et-al-supervised-contextpred-sup-cp"
      ],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "evaluation",
      "reporting_gaps": [
        "MoleculeNet version is not reported.",
        "Realized test-set sample sizes are not reported.",
        "Repeat count is not stated for Table 2.",
        "Exact checkpoint revision and release date are not reported.",
        "Scaffold splitting is the within-dataset split protocol, not the method used to select the eight benchmark datasets.",
        "Table rows vary by adaptation strategy; those strategies are not normalized here as standalone model identities.",
        "benchmark version",
        "numeric result"
      ],
      "scope": {
        "n": 8,
        "reporting_status": "reported",
        "selection": "filtered",
        "selection_method": "Eight MoleculeNet classification datasets: BBBP, Tox21, ToxCast, SIDER, ClinTox, MUV, HIV, and BACE.",
        "subset_id": null,
        "subset_kind": "paper-specific",
        "type": "subset"
      },
      "status": "partial",
      "verification": {
        "last_verified": "2026-08-02",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "enhancing-molecular-property-prediction-with-auxiliary",
      "work_version_id": "enhancing-molecular-property-prediction-with-auxiliary-pmc-version-1"
    },
    {
      "benchmark_id": "moleculenet",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-08-02",
          "id": "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-4-use-evidence-1",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "e4a0f03a9138ea108312ece37c02d43c87b45f96191019b6c6d8661dcb53ea57",
            "type": "table",
            "value": "Table 3"
          },
          "source_id": "enhancing-molecular-property-prediction-with-auxiliary",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-4-use-evidence-2",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "d980d38df62519335682b1964134b0b656d479c9f7c490e8736c48b95492f4ea",
            "type": "section",
            "value": "Results and discussion > Experimental setup"
          },
          "source_id": "enhancing-molecular-property-prediction-with-auxiliary",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-4-use-evidence-3",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "d980d38df62519335682b1964134b0b656d479c9f7c490e8736c48b95492f4ea",
            "type": "section",
            "value": "Results and discussion > Experimental setup"
          },
          "source_id": "enhancing-molecular-property-prediction-with-auxiliary",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-4-use-evidence-4",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "d980d38df62519335682b1964134b0b656d479c9f7c490e8736c48b95492f4ea",
            "type": "section",
            "value": "Results and discussion > Experimental setup"
          },
          "source_id": "enhancing-molecular-property-prediction-with-auxiliary",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-4-use-evidence-5",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "1be2ace855d6b72411fc8616ecadc846eaa3a9bd399ec7f33e242b91fe899d13",
            "type": "table",
            "value": "Table 3, dataset columns"
          },
          "source_id": "enhancing-molecular-property-prediction-with-auxiliary",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-4-use-evidence-6",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "8e19f2f648510df76c5f9b958ebd2e94207d5a8eb90aa81c8b47987fb98a28ed",
            "type": "section",
            "value": "Results and discussion > Reproducibility and implementation details"
          },
          "source_id": "enhancing-molecular-property-prediction-with-auxiliary",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-4-use-evidence-7",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "a03c4558af1a1f6d24f6feec8d955744b46edecf210c5ef1d0f146cbb1250bfc",
            "type": "other",
            "value": "Experimental setup; reference 17"
          },
          "source_id": "enhancing-molecular-property-prediction-with-auxiliary",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-4-use-evidence-8",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "e1a9227d045fe463a911bb9aa43afc5dd42c9155ced19e843543824d925eedc5",
            "type": "table",
            "value": "Table 3 caption and footnote"
          },
          "source_id": "enhancing-molecular-property-prediction-with-auxiliary",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        }
      ],
      "id": "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-4-use",
      "metric_labels": [
        "Test ROC-AUC (Sup; AM,CP,EP,IG,MP)"
      ],
      "model_ids": [
        "hu-et-al-supervised-sup"
      ],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "evaluation",
      "reporting_gaps": [
        "MoleculeNet version is not reported.",
        "Realized test-set sample sizes are not reported.",
        "Repeat count is not stated for Table 3.",
        "Exact checkpoint revision and release date are not reported.",
        "Scaffold splitting is the within-dataset split protocol, not the method used to select the eight benchmark datasets.",
        "Table rows vary by adaptation strategy; those strategies are not normalized here as standalone model identities.",
        "benchmark version",
        "numeric result"
      ],
      "scope": {
        "n": 8,
        "reporting_status": "reported",
        "selection": "filtered",
        "selection_method": "Eight MoleculeNet classification datasets: BBBP, Tox21, ToxCast, SIDER, ClinTox, MUV, HIV, and BACE.",
        "subset_id": null,
        "subset_kind": "paper-specific",
        "type": "subset"
      },
      "status": "partial",
      "verification": {
        "last_verified": "2026-08-02",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "enhancing-molecular-property-prediction-with-auxiliary",
      "work_version_id": "enhancing-molecular-property-prediction-with-auxiliary-pmc-version-1"
    },
    {
      "benchmark_id": "ab-bind",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ab-bind-5-use-evidence-1",
          "locator": {
            "document_page": 9,
            "note": null,
            "printed_page": "9",
            "source_fragment_sha256": "dc44c98e93c8c3d67efeb41b1dbdce831c1b8b028ceddcb26568283cd41be29d",
            "type": "section",
            "value": "2.3"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ab-bind-5-use-evidence-2",
          "locator": {
            "document_page": 9,
            "note": null,
            "printed_page": "9",
            "source_fragment_sha256": "f9b2c4f98ae525af2278b1f95e7f3d336312a87e1cfb0906bff4f6ee3c9badb5",
            "type": "section",
            "value": "2.3"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ab-bind-5-use-evidence-3",
          "locator": {
            "document_page": 9,
            "note": null,
            "printed_page": "9",
            "source_fragment_sha256": "ff076448cdbf882041ddebbcf05239028ae82fec4667f21b8b6a81a9a1e55480",
            "type": "section",
            "value": "2.3"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ab-bind-5-use-evidence-4",
          "locator": {
            "document_page": 9,
            "note": null,
            "printed_page": "9",
            "source_fragment_sha256": "237c15e0c4897fd4a1934eb9f4f79941ed3655b0b9161b7f8d8097cfbcb544c6",
            "type": "section",
            "value": "2.3"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ab-bind-5-use-evidence-5",
          "locator": {
            "document_page": 10,
            "note": null,
            "printed_page": "10",
            "source_fragment_sha256": "77f8990bd246ac794e676ea8630c0611d9ef4b6c32039e704ec0868e538e0a3b",
            "type": "table",
            "value": "Table 1"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ab-bind-5-use-evidence-6",
          "locator": {
            "document_page": 9,
            "note": null,
            "printed_page": "9",
            "source_fragment_sha256": "933d231cd432285faacc31ed1351cebd2a676f1687d17d3080be25a4a7a3682f",
            "type": "section",
            "value": "2.3"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ab-bind-5-use-evidence-7",
          "locator": {
            "document_page": 10,
            "note": null,
            "printed_page": "10",
            "source_fragment_sha256": "77f8990bd246ac794e676ea8630c0611d9ef4b6c32039e704ec0868e538e0a3b",
            "type": "table",
            "value": "Table 1"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        }
      ],
      "id": "explainable-protein-protein-binding-affinity-predictio-ab-bind-5-use",
      "metric_labels": [],
      "model_ids": [
        "not-reported-balm-ppi"
      ],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "evaluation",
      "reporting_gaps": [
        "Benchmark version is not reported.",
        "BALM-PPI provider and version are not explicitly reported.",
        "benchmark version",
        "metric",
        "numeric result"
      ],
      "scope": {
        "n": 272,
        "reporting_status": "reported",
        "selection": "filtered",
        "selection_method": "Removed parent-PDB overlap or identical antigen-antibody sequence-pair overlap with PPB-Affinity training data",
        "subset_id": "ppb-affinity-decontaminated-ab-bind",
        "subset_kind": "paper-specific",
        "type": "subset"
      },
      "status": "partial",
      "verification": {
        "last_verified": "2026-08-20",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "explainable-protein-protein-binding-affinity-predictio",
      "work_version_id": "explainable-protein-protein-binding-affinity-predictio-biorxiv-version-posted-2"
    },
    {
      "benchmark_id": "ab-bind",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ab-bind-6-use-evidence-1",
          "locator": {
            "document_page": 10,
            "note": null,
            "printed_page": "10",
            "source_fragment_sha256": "277a781f1d4d64c3ef7c77a1c5f08bc68e8613b5f709b3236e1ded422b8a1a43",
            "type": "table",
            "value": "Table 1"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ab-bind-6-use-evidence-2",
          "locator": {
            "document_page": 10,
            "note": null,
            "printed_page": "10",
            "source_fragment_sha256": "77f8990bd246ac794e676ea8630c0611d9ef4b6c32039e704ec0868e538e0a3b",
            "type": "table",
            "value": "Table 1"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ab-bind-6-use-evidence-3",
          "locator": {
            "document_page": 10,
            "note": null,
            "printed_page": "10",
            "source_fragment_sha256": "1b91828f974579c2bb45ae00da6f3b986f4d9a29c9e4068e6d0076962a0359d0",
            "type": "table",
            "value": "Table 1"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ab-bind-6-use-evidence-4",
          "locator": {
            "document_page": 10,
            "note": null,
            "printed_page": "10",
            "source_fragment_sha256": "77f8990bd246ac794e676ea8630c0611d9ef4b6c32039e704ec0868e538e0a3b",
            "type": "table",
            "value": "Table 1"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ab-bind-6-use-evidence-5",
          "locator": {
            "document_page": 10,
            "note": null,
            "printed_page": "10",
            "source_fragment_sha256": "1b91828f974579c2bb45ae00da6f3b986f4d9a29c9e4068e6d0076962a0359d0",
            "type": "table",
            "value": "Table 1"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ab-bind-6-use-evidence-6",
          "locator": {
            "document_page": 10,
            "note": null,
            "printed_page": "10",
            "source_fragment_sha256": "277a781f1d4d64c3ef7c77a1c5f08bc68e8613b5f709b3236e1ded422b8a1a43",
            "type": "table",
            "value": "Table 1"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        }
      ],
      "id": "explainable-protein-protein-binding-affinity-predictio-ab-bind-6-use",
      "metric_labels": [],
      "model_ids": [
        "not-reported-balm-ppi"
      ],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "fine-tuning",
      "reporting_gaps": [
        "Benchmark version and exact realized training/evaluation n are not reported.",
        "BALM-PPI provider and version are not explicitly reported."
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_applicable",
        "selection": null,
        "selection_method": "Few-shot 10%",
        "subset_id": null,
        "subset_kind": "not-applicable",
        "type": "unknown"
      },
      "status": "non-evaluation",
      "verification": {
        "last_verified": "2026-08-20",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "explainable-protein-protein-binding-affinity-predictio",
      "work_version_id": "explainable-protein-protein-binding-affinity-predictio-biorxiv-version-posted-2"
    },
    {
      "benchmark_id": "ab-bind",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ab-bind-7-use-evidence-1",
          "locator": {
            "document_page": 10,
            "note": null,
            "printed_page": "10",
            "source_fragment_sha256": "277a781f1d4d64c3ef7c77a1c5f08bc68e8613b5f709b3236e1ded422b8a1a43",
            "type": "table",
            "value": "Table 1"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ab-bind-7-use-evidence-2",
          "locator": {
            "document_page": 10,
            "note": null,
            "printed_page": "10",
            "source_fragment_sha256": "77f8990bd246ac794e676ea8630c0611d9ef4b6c32039e704ec0868e538e0a3b",
            "type": "table",
            "value": "Table 1"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ab-bind-7-use-evidence-3",
          "locator": {
            "document_page": 10,
            "note": null,
            "printed_page": "10",
            "source_fragment_sha256": "1e8ea7b7554b30fbdb659f7bab40b75ff5a3fc71122c16e23764a091c2561dde",
            "type": "table",
            "value": "Table 1"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ab-bind-7-use-evidence-4",
          "locator": {
            "document_page": 10,
            "note": null,
            "printed_page": "10",
            "source_fragment_sha256": "77f8990bd246ac794e676ea8630c0611d9ef4b6c32039e704ec0868e538e0a3b",
            "type": "table",
            "value": "Table 1"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ab-bind-7-use-evidence-5",
          "locator": {
            "document_page": 10,
            "note": null,
            "printed_page": "10",
            "source_fragment_sha256": "1e8ea7b7554b30fbdb659f7bab40b75ff5a3fc71122c16e23764a091c2561dde",
            "type": "table",
            "value": "Table 1"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ab-bind-7-use-evidence-6",
          "locator": {
            "document_page": 10,
            "note": null,
            "printed_page": "10",
            "source_fragment_sha256": "277a781f1d4d64c3ef7c77a1c5f08bc68e8613b5f709b3236e1ded422b8a1a43",
            "type": "table",
            "value": "Table 1"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        }
      ],
      "id": "explainable-protein-protein-binding-affinity-predictio-ab-bind-7-use",
      "metric_labels": [],
      "model_ids": [
        "not-reported-balm-ppi"
      ],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "fine-tuning",
      "reporting_gaps": [
        "Benchmark version and exact realized training/evaluation n are not reported.",
        "BALM-PPI provider and version are not explicitly reported."
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_applicable",
        "selection": null,
        "selection_method": "Few-shot 20%",
        "subset_id": null,
        "subset_kind": "not-applicable",
        "type": "unknown"
      },
      "status": "non-evaluation",
      "verification": {
        "last_verified": "2026-08-20",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "explainable-protein-protein-binding-affinity-predictio",
      "work_version_id": "explainable-protein-protein-binding-affinity-predictio-biorxiv-version-posted-2"
    },
    {
      "benchmark_id": "ab-bind",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ab-bind-8-use-evidence-1",
          "locator": {
            "document_page": 10,
            "note": null,
            "printed_page": "10",
            "source_fragment_sha256": "277a781f1d4d64c3ef7c77a1c5f08bc68e8613b5f709b3236e1ded422b8a1a43",
            "type": "table",
            "value": "Table 1"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ab-bind-8-use-evidence-2",
          "locator": {
            "document_page": 10,
            "note": null,
            "printed_page": "10",
            "source_fragment_sha256": "77f8990bd246ac794e676ea8630c0611d9ef4b6c32039e704ec0868e538e0a3b",
            "type": "table",
            "value": "Table 1"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ab-bind-8-use-evidence-3",
          "locator": {
            "document_page": 10,
            "note": null,
            "printed_page": "10",
            "source_fragment_sha256": "c756ada5b3288d796d4039502eab8138ca6beab807ec998ea555ef672fa12fe3",
            "type": "table",
            "value": "Table 1"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ab-bind-8-use-evidence-4",
          "locator": {
            "document_page": 10,
            "note": null,
            "printed_page": "10",
            "source_fragment_sha256": "77f8990bd246ac794e676ea8630c0611d9ef4b6c32039e704ec0868e538e0a3b",
            "type": "table",
            "value": "Table 1"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ab-bind-8-use-evidence-5",
          "locator": {
            "document_page": 10,
            "note": null,
            "printed_page": "10",
            "source_fragment_sha256": "c756ada5b3288d796d4039502eab8138ca6beab807ec998ea555ef672fa12fe3",
            "type": "table",
            "value": "Table 1"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ab-bind-8-use-evidence-6",
          "locator": {
            "document_page": 10,
            "note": null,
            "printed_page": "10",
            "source_fragment_sha256": "277a781f1d4d64c3ef7c77a1c5f08bc68e8613b5f709b3236e1ded422b8a1a43",
            "type": "table",
            "value": "Table 1"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        }
      ],
      "id": "explainable-protein-protein-binding-affinity-predictio-ab-bind-8-use",
      "metric_labels": [],
      "model_ids": [
        "not-reported-balm-ppi"
      ],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "fine-tuning",
      "reporting_gaps": [
        "Benchmark version and exact realized training/evaluation n are not reported.",
        "BALM-PPI provider and version are not explicitly reported."
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_applicable",
        "selection": null,
        "selection_method": "Few-shot 30%",
        "subset_id": null,
        "subset_kind": "not-applicable",
        "type": "unknown"
      },
      "status": "non-evaluation",
      "verification": {
        "last_verified": "2026-08-20",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "explainable-protein-protein-binding-affinity-predictio",
      "work_version_id": "explainable-protein-protein-binding-affinity-predictio-biorxiv-version-posted-2"
    },
    {
      "benchmark_id": "abbibench",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-abbibench-10-use-evidence-1",
          "locator": {
            "document_page": 40,
            "note": null,
            "printed_page": "40",
            "source_fragment_sha256": "8ea8213a04d36e2c1cd242d98510c4285357056e1c386cad23a14485e199c0d1",
            "type": "table",
            "value": "Table S6"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-abbibench-10-use-evidence-2",
          "locator": {
            "document_page": 40,
            "note": null,
            "printed_page": "40",
            "source_fragment_sha256": "40a3587fbca1336904fbe1faf2fdd335b8d70ed1960e13d2bffa438202539fdb",
            "type": "table",
            "value": "Table S6"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-abbibench-10-use-evidence-3",
          "locator": {
            "document_page": 10,
            "note": null,
            "printed_page": "10",
            "source_fragment_sha256": "374ac53fb1b979b6e52e07b1bb1feaddb3b7f86880a030bc3f092a7b505a64a2",
            "type": "section",
            "value": "2.4"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-abbibench-10-use-evidence-4",
          "locator": {
            "document_page": 10,
            "note": null,
            "printed_page": "10",
            "source_fragment_sha256": "a939f25ea6673c6e74253917c990fa19d3dd9d3b78d70c0e0bd51eeb1966838a",
            "type": "section",
            "value": "2.4"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-abbibench-10-use-evidence-5",
          "locator": {
            "document_page": 39,
            "note": null,
            "printed_page": "39",
            "source_fragment_sha256": "a61f46da8c9072a1f3a0552d849679a8889152ca2c52aff19fc6f86883061d91",
            "type": "table",
            "value": "Table S5"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-abbibench-10-use-evidence-6",
          "locator": {
            "document_page": 40,
            "note": null,
            "printed_page": "40",
            "source_fragment_sha256": "768372bc7969e9466bf652abc7059f0b4f59d006d13108c1b31043ef15dc6d96",
            "type": "table",
            "value": "Table S6"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-abbibench-10-use-evidence-7",
          "locator": {
            "document_page": 40,
            "note": null,
            "printed_page": "40",
            "source_fragment_sha256": "92fa133150976f017e3e8a68b45443ba16571e8011c86fbc018299501966a362",
            "type": "table",
            "value": "Table S6"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-abbibench-10-use-evidence-8",
          "locator": {
            "document_page": 40,
            "note": null,
            "printed_page": "40",
            "source_fragment_sha256": "2ac58a3e348dcfb40850947c4ac956b88c1a31736e6f3a9e103f11f4efb39413",
            "type": "table",
            "value": "Table S6"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        }
      ],
      "id": "explainable-protein-protein-binding-affinity-predictio-abbibench-10-use",
      "metric_labels": [
        "RMSE"
      ],
      "model_ids": [
        "not-reported-balm-ppi"
      ],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "fine-tuning",
      "reporting_gaps": [
        "Benchmark version and exact realized training/evaluation n per assay are not reported.",
        "BALM-PPI provider and version are not explicitly reported."
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_applicable",
        "selection": null,
        "selection_method": "Few-Shot 10%",
        "subset_id": null,
        "subset_kind": "not-applicable",
        "type": "unknown"
      },
      "status": "non-evaluation",
      "verification": {
        "last_verified": "2026-08-20",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "explainable-protein-protein-binding-affinity-predictio",
      "work_version_id": "explainable-protein-protein-binding-affinity-predictio-biorxiv-version-posted-2"
    },
    {
      "benchmark_id": "abbibench",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-abbibench-11-use-evidence-1",
          "locator": {
            "document_page": 40,
            "note": null,
            "printed_page": "40",
            "source_fragment_sha256": "8ea8213a04d36e2c1cd242d98510c4285357056e1c386cad23a14485e199c0d1",
            "type": "table",
            "value": "Table S6"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-abbibench-11-use-evidence-2",
          "locator": {
            "document_page": 40,
            "note": null,
            "printed_page": "40",
            "source_fragment_sha256": "40a3587fbca1336904fbe1faf2fdd335b8d70ed1960e13d2bffa438202539fdb",
            "type": "table",
            "value": "Table S6"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-abbibench-11-use-evidence-3",
          "locator": {
            "document_page": 10,
            "note": null,
            "printed_page": "10",
            "source_fragment_sha256": "374ac53fb1b979b6e52e07b1bb1feaddb3b7f86880a030bc3f092a7b505a64a2",
            "type": "section",
            "value": "2.4"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-abbibench-11-use-evidence-4",
          "locator": {
            "document_page": 10,
            "note": null,
            "printed_page": "10",
            "source_fragment_sha256": "a939f25ea6673c6e74253917c990fa19d3dd9d3b78d70c0e0bd51eeb1966838a",
            "type": "section",
            "value": "2.4"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-abbibench-11-use-evidence-5",
          "locator": {
            "document_page": 39,
            "note": null,
            "printed_page": "39",
            "source_fragment_sha256": "a61f46da8c9072a1f3a0552d849679a8889152ca2c52aff19fc6f86883061d91",
            "type": "table",
            "value": "Table S5"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-abbibench-11-use-evidence-6",
          "locator": {
            "document_page": 40,
            "note": null,
            "printed_page": "40",
            "source_fragment_sha256": "8ddf3689ba4571621e8c9ef739469a0587f788bb219b953c1d583e2820561d8b",
            "type": "table",
            "value": "Table S6"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-abbibench-11-use-evidence-7",
          "locator": {
            "document_page": 40,
            "note": null,
            "printed_page": "40",
            "source_fragment_sha256": "92fa133150976f017e3e8a68b45443ba16571e8011c86fbc018299501966a362",
            "type": "table",
            "value": "Table S6"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-abbibench-11-use-evidence-8",
          "locator": {
            "document_page": 40,
            "note": null,
            "printed_page": "40",
            "source_fragment_sha256": "2ac58a3e348dcfb40850947c4ac956b88c1a31736e6f3a9e103f11f4efb39413",
            "type": "table",
            "value": "Table S6"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        }
      ],
      "id": "explainable-protein-protein-binding-affinity-predictio-abbibench-11-use",
      "metric_labels": [
        "RMSE"
      ],
      "model_ids": [
        "not-reported-balm-ppi"
      ],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "fine-tuning",
      "reporting_gaps": [
        "Benchmark version and exact realized training/evaluation n per assay are not reported.",
        "BALM-PPI provider and version are not explicitly reported."
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_applicable",
        "selection": null,
        "selection_method": "Few-Shot 20%",
        "subset_id": null,
        "subset_kind": "not-applicable",
        "type": "unknown"
      },
      "status": "non-evaluation",
      "verification": {
        "last_verified": "2026-08-20",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "explainable-protein-protein-binding-affinity-predictio",
      "work_version_id": "explainable-protein-protein-binding-affinity-predictio-biorxiv-version-posted-2"
    },
    {
      "benchmark_id": "abbibench",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-abbibench-12-use-evidence-1",
          "locator": {
            "document_page": 40,
            "note": null,
            "printed_page": "40",
            "source_fragment_sha256": "8ea8213a04d36e2c1cd242d98510c4285357056e1c386cad23a14485e199c0d1",
            "type": "table",
            "value": "Table S6"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-abbibench-12-use-evidence-2",
          "locator": {
            "document_page": 40,
            "note": null,
            "printed_page": "40",
            "source_fragment_sha256": "40a3587fbca1336904fbe1faf2fdd335b8d70ed1960e13d2bffa438202539fdb",
            "type": "table",
            "value": "Table S6"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-abbibench-12-use-evidence-3",
          "locator": {
            "document_page": 10,
            "note": null,
            "printed_page": "10",
            "source_fragment_sha256": "374ac53fb1b979b6e52e07b1bb1feaddb3b7f86880a030bc3f092a7b505a64a2",
            "type": "section",
            "value": "2.4"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-abbibench-12-use-evidence-4",
          "locator": {
            "document_page": 10,
            "note": null,
            "printed_page": "10",
            "source_fragment_sha256": "a939f25ea6673c6e74253917c990fa19d3dd9d3b78d70c0e0bd51eeb1966838a",
            "type": "section",
            "value": "2.4"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-abbibench-12-use-evidence-5",
          "locator": {
            "document_page": 39,
            "note": null,
            "printed_page": "39",
            "source_fragment_sha256": "a61f46da8c9072a1f3a0552d849679a8889152ca2c52aff19fc6f86883061d91",
            "type": "table",
            "value": "Table S5"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-abbibench-12-use-evidence-6",
          "locator": {
            "document_page": 40,
            "note": null,
            "printed_page": "40",
            "source_fragment_sha256": "28ff88169b22a9b38b1b08f6c8aba5a8260a76686ff5cbbe0a8cc5c486cbe987",
            "type": "table",
            "value": "Table S6"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-abbibench-12-use-evidence-7",
          "locator": {
            "document_page": 40,
            "note": null,
            "printed_page": "40",
            "source_fragment_sha256": "92fa133150976f017e3e8a68b45443ba16571e8011c86fbc018299501966a362",
            "type": "table",
            "value": "Table S6"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-abbibench-12-use-evidence-8",
          "locator": {
            "document_page": 40,
            "note": null,
            "printed_page": "40",
            "source_fragment_sha256": "2ac58a3e348dcfb40850947c4ac956b88c1a31736e6f3a9e103f11f4efb39413",
            "type": "table",
            "value": "Table S6"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        }
      ],
      "id": "explainable-protein-protein-binding-affinity-predictio-abbibench-12-use",
      "metric_labels": [
        "RMSE"
      ],
      "model_ids": [
        "not-reported-balm-ppi"
      ],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "fine-tuning",
      "reporting_gaps": [
        "Benchmark version and exact realized training/evaluation n per assay are not reported.",
        "BALM-PPI provider and version are not explicitly reported."
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_applicable",
        "selection": null,
        "selection_method": "Few-Shot 30%",
        "subset_id": null,
        "subset_kind": "not-applicable",
        "type": "unknown"
      },
      "status": "non-evaluation",
      "verification": {
        "last_verified": "2026-08-20",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "explainable-protein-protein-binding-affinity-predictio",
      "work_version_id": "explainable-protein-protein-binding-affinity-predictio-biorxiv-version-posted-2"
    },
    {
      "benchmark_id": "abbibench",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-abbibench-9-use-evidence-1",
          "locator": {
            "document_page": 10,
            "note": null,
            "printed_page": "10",
            "source_fragment_sha256": "ac7f39271336ac340dc641ce0cbd89e85c7304d68dcd4723dd2979d5cff5e879",
            "type": "section",
            "value": "2.4"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-abbibench-9-use-evidence-2",
          "locator": {
            "document_page": 10,
            "note": null,
            "printed_page": "10",
            "source_fragment_sha256": "374ac53fb1b979b6e52e07b1bb1feaddb3b7f86880a030bc3f092a7b505a64a2",
            "type": "section",
            "value": "2.4"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-abbibench-9-use-evidence-3",
          "locator": {
            "document_page": 10,
            "note": null,
            "printed_page": "10",
            "source_fragment_sha256": "538e34f37e49d9978d8454e454caff9d11415bc0b5302cf5e54718e274ba7bdb",
            "type": "section",
            "value": "2.4"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-abbibench-9-use-evidence-4",
          "locator": {
            "document_page": 10,
            "note": null,
            "printed_page": "10",
            "source_fragment_sha256": "a939f25ea6673c6e74253917c990fa19d3dd9d3b78d70c0e0bd51eeb1966838a",
            "type": "section",
            "value": "2.4"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-abbibench-9-use-evidence-5",
          "locator": {
            "document_page": 39,
            "note": null,
            "printed_page": "39",
            "source_fragment_sha256": "a61f46da8c9072a1f3a0552d849679a8889152ca2c52aff19fc6f86883061d91",
            "type": "table",
            "value": "Table S5"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-abbibench-9-use-evidence-6",
          "locator": {
            "document_page": 39,
            "note": null,
            "printed_page": "39",
            "source_fragment_sha256": "aa17e6628b8d28a392b2cf93469fb2de6c561bfb66d7d372880be07b21a1e03d",
            "type": "table",
            "value": "Table S5"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-abbibench-9-use-evidence-7",
          "locator": {
            "document_page": 40,
            "note": null,
            "printed_page": "40",
            "source_fragment_sha256": "92fa133150976f017e3e8a68b45443ba16571e8011c86fbc018299501966a362",
            "type": "table",
            "value": "Table S6"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-abbibench-9-use-evidence-8",
          "locator": {
            "document_page": 40,
            "note": null,
            "printed_page": "40",
            "source_fragment_sha256": "56599cfe1d7114e4870a3c6d96659929b388081f63a5e63276fb7d29e65a993e",
            "type": "table",
            "value": "Table S6"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        }
      ],
      "id": "explainable-protein-protein-binding-affinity-predictio-abbibench-9-use",
      "metric_labels": [
        "RMSE"
      ],
      "model_ids": [
        "not-reported-balm-ppi"
      ],
      "notes": "AI-assisted double-pass extraction; n=9 counts selected assay tracks, not individual examples. Values are limited to independently supported claims.",
      "relation_type": "evaluation",
      "reporting_gaps": [
        "Benchmark version is not reported.",
        "The source reports assay inventories, but not an exact realized evaluation n for every assay.",
        "BALM-PPI provider and version are not explicitly reported.",
        "benchmark version"
      ],
      "scope": {
        "n": 9,
        "reporting_status": "reported",
        "selection": "filtered",
        "selection_method": "Nine selected DMS assays: 1n8z, 1mhp_LC, 3gbn_h1, 3gbn_h9, 4fqi_h1, aayl49_ml, aayl50_LC, aayl51, aayl52",
        "subset_id": "selected-nine-dms-assays",
        "subset_kind": "paper-specific",
        "type": "subset"
      },
      "status": "partial",
      "verification": {
        "last_verified": "2026-08-20",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "explainable-protein-protein-binding-affinity-predictio",
      "work_version_id": "explainable-protein-protein-binding-affinity-predictio-biorxiv-version-posted-2"
    },
    {
      "benchmark_id": "ppb-affinity",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-1-use-evidence-1",
          "locator": {
            "document_page": 4,
            "note": null,
            "printed_page": "4",
            "source_fragment_sha256": "251e54038766301f859f943db8c223378dff2459f26182fff9c1572dc10117ff",
            "type": "section",
            "value": "2.1"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-1-use-evidence-2",
          "locator": {
            "document_page": 4,
            "note": null,
            "printed_page": "4",
            "source_fragment_sha256": "251e54038766301f859f943db8c223378dff2459f26182fff9c1572dc10117ff",
            "type": "section",
            "value": "2.1"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-1-use-evidence-3",
          "locator": {
            "document_page": 22,
            "note": null,
            "printed_page": "22",
            "source_fragment_sha256": "09679784e0eb08ac64c213a0c0f6854e62777f8d16f333a078eef52c0e9033e4",
            "type": "section",
            "value": "4.1 Dataset"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-1-use-evidence-4",
          "locator": {
            "document_page": 22,
            "note": null,
            "printed_page": "22",
            "source_fragment_sha256": "09679784e0eb08ac64c213a0c0f6854e62777f8d16f333a078eef52c0e9033e4",
            "type": "section",
            "value": "4.1 Dataset"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-1-use-evidence-5",
          "locator": {
            "document_page": 37,
            "note": null,
            "printed_page": "37",
            "source_fragment_sha256": "c0c2a93a6847dc0927379c788bfdee376d2145fd125ab528a0eab728b5c0ec6b",
            "type": "table",
            "value": "Table S1"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-1-use-evidence-6",
          "locator": {
            "document_page": 22,
            "note": null,
            "printed_page": "22",
            "source_fragment_sha256": "09679784e0eb08ac64c213a0c0f6854e62777f8d16f333a078eef52c0e9033e4",
            "type": "section",
            "value": "4.1 Dataset"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-1-use-evidence-7",
          "locator": {
            "document_page": 4,
            "note": null,
            "printed_page": "4",
            "source_fragment_sha256": "251e54038766301f859f943db8c223378dff2459f26182fff9c1572dc10117ff",
            "type": "section",
            "value": "2.1"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        }
      ],
      "id": "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-1-use",
      "metric_labels": [],
      "model_ids": [
        "not-reported-balm-ppi"
      ],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "training",
      "reporting_gaps": [
        "Benchmark version is not reported.",
        "Exact per-fold training sizes are not reported for cold and sequence-similarity splits.",
        "BALM-PPI provider and version are not explicitly reported."
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_applicable",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-applicable",
        "type": "unknown"
      },
      "status": "non-evaluation",
      "verification": {
        "last_verified": "2026-08-20",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "explainable-protein-protein-binding-affinity-predictio",
      "work_version_id": "explainable-protein-protein-binding-affinity-predictio-biorxiv-version-posted-2"
    },
    {
      "benchmark_id": "ppb-affinity",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-2-use-evidence-1",
          "locator": {
            "document_page": 37,
            "note": null,
            "printed_page": "37",
            "source_fragment_sha256": "486e03dd8eed8ae7b5509e21ba3a10cf94caed78674d36f6ad64d8cdb4be9502",
            "type": "table",
            "value": "Table S2"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-2-use-evidence-2",
          "locator": {
            "document_page": 37,
            "note": null,
            "printed_page": "37",
            "source_fragment_sha256": "c0c2a93a6847dc0927379c788bfdee376d2145fd125ab528a0eab728b5c0ec6b",
            "type": "table",
            "value": "Table S1"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-2-use-evidence-3",
          "locator": {
            "document_page": 23,
            "note": null,
            "printed_page": "23",
            "source_fragment_sha256": "cb079fe4e4e404acb1a700006b5512f35d5f00d09ae061aadce5b64f3eb88642",
            "type": "section",
            "value": "4.2"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-2-use-evidence-4",
          "locator": {
            "document_page": 23,
            "note": null,
            "printed_page": "23",
            "source_fragment_sha256": "98174ffbdffd93a2e5f5abc0227c57bc8669370e0ef47e17373746ebcbf18fe8",
            "type": "section",
            "value": "4.2"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-2-use-evidence-5",
          "locator": {
            "document_page": 37,
            "note": null,
            "printed_page": "37",
            "source_fragment_sha256": "909e265166668205140e4aca9dfd078c291f452910aef314c48bf9667cf1cc93",
            "type": "table",
            "value": "Table S2"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-2-use-evidence-6",
          "locator": {
            "document_page": 37,
            "note": null,
            "printed_page": "37",
            "source_fragment_sha256": "f10a5838451aa67ba568054144d64f2a0bf098bc2bd019ad1a8bc89895838e0d",
            "type": "table",
            "value": "Table S2"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-2-use-evidence-7",
          "locator": {
            "document_page": 37,
            "note": null,
            "printed_page": "37",
            "source_fragment_sha256": "5c3db1995e35f4c6cc65b286cfbcdf3014346e2d75ecc8a99ddad6f43323ad84",
            "type": "table",
            "value": "Table S2"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-2-use-evidence-8",
          "locator": {
            "document_page": 37,
            "note": null,
            "printed_page": "37",
            "source_fragment_sha256": "f7edea70ddad8dde391ebf1ecc1d74e801567f4d29419fd41372103e6062c800",
            "type": "table",
            "value": "Table S2"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        }
      ],
      "id": "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-2-use",
      "metric_labels": [
        "RMSE"
      ],
      "model_ids": [
        "not-reported-balm-ppi",
        "not-reported-balm-ppi-without-peft",
        "not-reported-balm-ppi-standard-regression-baseline"
      ],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "evaluation",
      "reporting_gaps": [
        "Benchmark version is not reported.",
        "The exact realized random-split test n is not reported; only approximately 2,404 per fold is stated.",
        "Model providers and versions are not explicitly reported.",
        "benchmark version",
        "realized n/scope"
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_reported",
        "selection": null,
        "selection_method": "Random five-fold cross-validation with 80% training and 20% testing per fold",
        "subset_id": null,
        "subset_kind": "formal-subset",
        "type": "subset"
      },
      "status": "partial",
      "verification": {
        "last_verified": "2026-08-20",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "explainable-protein-protein-binding-affinity-predictio",
      "work_version_id": "explainable-protein-protein-binding-affinity-predictio-biorxiv-version-posted-2"
    },
    {
      "benchmark_id": "ppb-affinity",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-3-use-evidence-1",
          "locator": {
            "document_page": 37,
            "note": null,
            "printed_page": "37",
            "source_fragment_sha256": "486e03dd8eed8ae7b5509e21ba3a10cf94caed78674d36f6ad64d8cdb4be9502",
            "type": "table",
            "value": "Table S2"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-3-use-evidence-2",
          "locator": {
            "document_page": 7,
            "note": null,
            "printed_page": "7",
            "source_fragment_sha256": "6dd3c0fcdee91c365bf7197459c1ce9078b31837dcb9ff6bcb981c29c05677eb",
            "type": "section",
            "value": "2.2"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-3-use-evidence-3",
          "locator": {
            "document_page": 23,
            "note": null,
            "printed_page": "23",
            "source_fragment_sha256": "61a4fcff4e41635a7c101d8e403c9f9623821cb2c670cd8b1ae4ceec2ee97810",
            "type": "section",
            "value": "4.2"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-3-use-evidence-4",
          "locator": {
            "document_page": 23,
            "note": null,
            "printed_page": "23",
            "source_fragment_sha256": "a709fb0c702f04894e332141290acdbcb58b5ddb44f16f54d99e73a99cc7aae5",
            "type": "section",
            "value": "4.2"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-3-use-evidence-5",
          "locator": {
            "document_page": 37,
            "note": null,
            "printed_page": "37",
            "source_fragment_sha256": "e167bf1aded1d20111bd76accbc129b4512b53a0cf40e88a96cf3165a114fd32",
            "type": "table",
            "value": "Table S2"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-3-use-evidence-6",
          "locator": {
            "document_page": 37,
            "note": null,
            "printed_page": "37",
            "source_fragment_sha256": "85987473d4cb66e52cb37730360d67850746f8f7ca9f80b694e6ac59de41cc1a",
            "type": "table",
            "value": "Table S2"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-3-use-evidence-7",
          "locator": {
            "document_page": 37,
            "note": null,
            "printed_page": "37",
            "source_fragment_sha256": "4d62e72d44f2efecf8f67d0d0fa818204f2979d123df0181cfdaf8f47115218c",
            "type": "table",
            "value": "Table S2"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-3-use-evidence-8",
          "locator": {
            "document_page": 37,
            "note": null,
            "printed_page": "37",
            "source_fragment_sha256": "f7edea70ddad8dde391ebf1ecc1d74e801567f4d29419fd41372103e6062c800",
            "type": "table",
            "value": "Table S2"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        }
      ],
      "id": "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-3-use",
      "metric_labels": [
        "RMSE"
      ],
      "model_ids": [
        "not-reported-balm-ppi",
        "not-reported-balm-ppi-without-peft",
        "not-reported-balm-ppi-standard-regression-baseline"
      ],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "evaluation",
      "reporting_gaps": [
        "Benchmark version and exact realized test n are not reported.",
        "Model providers and versions are not explicitly reported.",
        "benchmark version",
        "realized n/scope"
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_reported",
        "selection": null,
        "selection_method": "GroupKFold cold split grouping interactions by PDB ID",
        "subset_id": null,
        "subset_kind": "formal-subset",
        "type": "subset"
      },
      "status": "partial",
      "verification": {
        "last_verified": "2026-08-20",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "explainable-protein-protein-binding-affinity-predictio",
      "work_version_id": "explainable-protein-protein-binding-affinity-predictio-biorxiv-version-posted-2"
    },
    {
      "benchmark_id": "ppb-affinity",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-4-use-evidence-1",
          "locator": {
            "document_page": 37,
            "note": null,
            "printed_page": "37",
            "source_fragment_sha256": "486e03dd8eed8ae7b5509e21ba3a10cf94caed78674d36f6ad64d8cdb4be9502",
            "type": "table",
            "value": "Table S2"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-4-use-evidence-2",
          "locator": {
            "document_page": 4,
            "note": null,
            "printed_page": "4",
            "source_fragment_sha256": "251e54038766301f859f943db8c223378dff2459f26182fff9c1572dc10117ff",
            "type": "section",
            "value": "2.1"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-4-use-evidence-3",
          "locator": {
            "document_page": 23,
            "note": null,
            "printed_page": "23",
            "source_fragment_sha256": "4d83687f90a7c51d2ed5a4bd405db004735f39c6ab7a2f06585cdb8dd25d893e",
            "type": "section",
            "value": "4.2"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-4-use-evidence-4",
          "locator": {
            "document_page": 23,
            "note": null,
            "printed_page": "23",
            "source_fragment_sha256": "0664c3752b5c66de570426996f9a9699b31d2fd73f15ef917b8bd6d5c9f76978",
            "type": "section",
            "value": "4.2"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-4-use-evidence-5",
          "locator": {
            "document_page": 37,
            "note": null,
            "printed_page": "37",
            "source_fragment_sha256": "e7d5a5fdec8b5fc61d5ef430f13131a1d742a7fee617e3395a37c9b5776c24d5",
            "type": "table",
            "value": "Table S2"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-4-use-evidence-6",
          "locator": {
            "document_page": 37,
            "note": null,
            "printed_page": "37",
            "source_fragment_sha256": "a9aef44f54490ab20c126f04d83bf7bbd618402d311636ab363693eb2c03f679",
            "type": "table",
            "value": "Table S2"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-4-use-evidence-7",
          "locator": {
            "document_page": 37,
            "note": null,
            "printed_page": "37",
            "source_fragment_sha256": "0626e11e2b459dc79394a1dd067766daaf3a58b9e070e1c017329af14fd77175",
            "type": "table",
            "value": "Table S2"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-4-use-evidence-8",
          "locator": {
            "document_page": 37,
            "note": null,
            "printed_page": "37",
            "source_fragment_sha256": "f7edea70ddad8dde391ebf1ecc1d74e801567f4d29419fd41372103e6062c800",
            "type": "table",
            "value": "Table S2"
          },
          "source_id": "explainable-protein-protein-binding-affinity-predictio",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        }
      ],
      "id": "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-4-use",
      "metric_labels": [
        "RMSE"
      ],
      "model_ids": [
        "not-reported-balm-ppi",
        "not-reported-balm-ppi-without-peft",
        "not-reported-balm-ppi-standard-regression-baseline"
      ],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "evaluation",
      "reporting_gaps": [
        "Benchmark version and exact realized test n are not reported.",
        "Model providers and versions are not explicitly reported.",
        "benchmark version",
        "realized n/scope"
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_reported",
        "selection": null,
        "selection_method": "K-mer Jaccard clustering with K=3, 30% similarity threshold, and pair exclusion",
        "subset_id": null,
        "subset_kind": "formal-subset",
        "type": "subset"
      },
      "status": "partial",
      "verification": {
        "last_verified": "2026-08-20",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "explainable-protein-protein-binding-affinity-predictio",
      "work_version_id": "explainable-protein-protein-binding-affinity-predictio-biorxiv-version-posted-2"
    },
    {
      "benchmark_id": "ppb-affinity",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-29",
          "id": "ppb-affinity-protein-protein-binding-affinity-dataset-ppb-affinity-1-use-evidence-1",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "8ed5cd9f2ec8f3e99003f86002c09d60429d00e009698b72ffb89807edf27d48",
            "type": "section",
            "value": "Background & Summary, paragraph 4"
          },
          "source_id": "ppb-affinity-protein-protein-binding-affinity-dataset",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "ppb-affinity-protein-protein-binding-affinity-dataset-ppb-affinity-1-use-evidence-2",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "45b00c34fea2a43873b0a36fcb8dd2f3963a44228c7b06d9f2b95ec8ed23a93b",
            "type": "section",
            "value": "Background & Summary, paragraph 4"
          },
          "source_id": "ppb-affinity-protein-protein-binding-affinity-dataset",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        }
      ],
      "id": "ppb-affinity-protein-protein-binding-affinity-dataset-ppb-affinity-1-use",
      "metric_labels": [],
      "model_ids": [],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "benchmark-creation",
      "reporting_gaps": [
        "No formal PPB-Affinity version label is printed.",
        "The exact dataset release date is not printed; the data citation reports only 2024.",
        "The licenses of the two GitHub repositories are not printed.",
        "Figure 3 raster content is unavailable locally, preventing inspection of any additional labeled partition values."
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_applicable",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-applicable",
        "type": "unknown"
      },
      "status": "non-evaluation",
      "verification": {
        "last_verified": "2026-07-29",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "ppb-affinity-protein-protein-binding-affinity-dataset",
      "work_version_id": "ppb-affinity-protein-protein-binding-affinity-dataset-2024-12-03"
    },
    {
      "benchmark_id": "ppb-affinity",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-29",
          "id": "ppb-affinity-protein-protein-binding-affinity-dataset-ppb-affinity-2-use-evidence-1",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "9a2259aaea58a47486805a4852f6e02fe1ad7c835590055205d5e8e4965732de",
            "type": "section",
            "value": "Technical Validation > Whole dataset, paragraph 47"
          },
          "source_id": "ppb-affinity-protein-protein-binding-affinity-dataset",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "ppb-affinity-protein-protein-binding-affinity-dataset-ppb-affinity-2-use-evidence-2",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "17f9b13ecd6db8e8230eb695628d10e9017024786c3d2069b6108b57a6e548f8",
            "type": "section",
            "value": "Abstract, paragraph 1"
          },
          "source_id": "ppb-affinity-protein-protein-binding-affinity-dataset",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "ppb-affinity-protein-protein-binding-affinity-dataset-ppb-affinity-2-use-evidence-3",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "39a68e7a01eb754e26480b4eb08a68babcf920d9974970ab232a73a2ba222d9e",
            "type": "section",
            "value": "Author contributions; Technical Validation > A benchmark affinity prediction, paragraph 46"
          },
          "source_id": "ppb-affinity-protein-protein-binding-affinity-dataset",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        }
      ],
      "id": "ppb-affinity-protein-protein-binding-affinity-dataset-ppb-affinity-2-use",
      "metric_labels": [],
      "model_ids": [
        "liu-et-al-benchmark-algorithm"
      ],
      "notes": "Owner-reviewed conservative publication: the creator evaluation is retained only as a partial relationship; conflicted settings and outcomes are omitted pending manual reconciliation.",
      "relation_type": "evaluation",
      "reporting_gaps": [
        "The testing-subset size and selection procedure are not printed in the available text.",
        "Repeats, seed, train-validation-test partition sizes, and model version are not reported in the available text.",
        "Linked Figures 4–5 are unavailable locally, so additional labeled subgroup and source-partition results could not be inspected.",
        "benchmark version",
        "realized n/scope",
        "metric",
        "numeric result",
        "prompt and tools",
        "grader and repeats"
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_reported",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-reported",
        "type": "unknown"
      },
      "status": "partial",
      "verification": {
        "last_verified": "2026-07-29",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "ppb-affinity-protein-protein-binding-affinity-dataset",
      "work_version_id": "ppb-affinity-protein-protein-binding-affinity-dataset-2024-12-03"
    },
    {
      "benchmark_id": "ppb-affinity",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-29",
          "id": "ppb-affinity-protein-protein-binding-affinity-dataset-ppb-affinity-3-use-evidence-1",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "14007219131d0fe4138757d6cd30cdbef9de103749253175d6b304c3debd5419",
            "type": "section",
            "value": "Technical Validation > Source datasets, paragraph 50"
          },
          "source_id": "ppb-affinity-protein-protein-binding-affinity-dataset",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "ppb-affinity-protein-protein-binding-affinity-dataset-ppb-affinity-3-use-evidence-2",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "be5619497d328b7999e87cf3244ef68d6b3af2a4e2831073c6c40082540321a9",
            "type": "section",
            "value": "Abstract, paragraph 1; Technical Validation > Source datasets, paragraph 50"
          },
          "source_id": "ppb-affinity-protein-protein-binding-affinity-dataset",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "ppb-affinity-protein-protein-binding-affinity-dataset-ppb-affinity-3-use-evidence-3",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "14007219131d0fe4138757d6cd30cdbef9de103749253175d6b304c3debd5419",
            "type": "section",
            "value": "Technical Validation > Source datasets, paragraph 50"
          },
          "source_id": "ppb-affinity-protein-protein-binding-affinity-dataset",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "ppb-affinity-protein-protein-binding-affinity-dataset-ppb-affinity-3-use-evidence-4",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "7be99f41eb6ae3bf869919bcad28b1c8e285282b4ccfd2b6c5a77c70da45bb22",
            "type": "section",
            "value": "Author contributions; Technical Validation > A benchmark affinity prediction, paragraph 46; Source datasets, paragraph 50"
          },
          "source_id": "ppb-affinity-protein-protein-binding-affinity-dataset",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        }
      ],
      "id": "ppb-affinity-protein-protein-binding-affinity-dataset-ppb-affinity-3-use",
      "metric_labels": [],
      "model_ids": [
        "liu-et-al-benchmark-algorithm"
      ],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "training",
      "reporting_gaps": [
        "The training percentages shown in Figure 5F cannot be inspected because the linked raster is unavailable locally.",
        "Training sample counts, split procedure, repeats, seed, hyperparameters, and model version are not reported in the available text."
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_applicable",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-applicable",
        "type": "unknown"
      },
      "status": "non-evaluation",
      "verification": {
        "last_verified": "2026-07-29",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "ppb-affinity-protein-protein-binding-affinity-dataset",
      "work_version_id": "ppb-affinity-protein-protein-binding-affinity-dataset-2024-12-03"
    },
    {
      "benchmark_id": "scbench",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-27",
          "id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an-scbench-1-use-evidence-1",
          "locator": {
            "document_page": 1,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "55c439997224d4eee49082aa94015e66229cae29a88c9e33f8c1f2ddc427c117",
            "type": "section",
            "value": "Abstract"
          },
          "source_id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-07-27",
          "id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an-scbench-1-use-evidence-2",
          "locator": {
            "document_page": 1,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "55c439997224d4eee49082aa94015e66229cae29a88c9e33f8c1f2ddc427c117",
            "type": "section",
            "value": "Abstract"
          },
          "source_id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        }
      ],
      "id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an-scbench-1-use",
      "metric_labels": [],
      "model_ids": [],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "benchmark-creation",
      "reporting_gaps": [
        "No explicit benchmark version is reported.",
        "The paper identifies the official repository but does not report its software license."
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_applicable",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-applicable",
        "type": "unknown"
      },
      "status": "non-evaluation",
      "verification": {
        "last_verified": "2026-07-27",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an",
      "work_version_id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an-arxiv-v1"
    },
    {
      "benchmark_id": "scbench",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-27",
          "id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an-scbench-2-use-evidence-1",
          "locator": {
            "document_page": 3,
            "note": null,
            "printed_page": "3",
            "source_fragment_sha256": "753123b3ed65c062dfd65259f9b239a5c6a4491578166bfa9faa6b7f496d1d98",
            "type": "table",
            "value": "Table 2"
          },
          "source_id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-07-27",
          "id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an-scbench-2-use-evidence-2",
          "locator": {
            "document_page": 3,
            "note": null,
            "printed_page": "3",
            "source_fragment_sha256": "753123b3ed65c062dfd65259f9b239a5c6a4491578166bfa9faa6b7f496d1d98",
            "type": "table",
            "value": "Table 2"
          },
          "source_id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        },
        {
          "accessed_date": "2026-07-27",
          "id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an-scbench-2-use-evidence-3",
          "locator": {
            "document_page": 3,
            "note": null,
            "printed_page": "3",
            "source_fragment_sha256": "2c44a1b144e18d454af3b6d83a33205199d080ab08e39fa74c2bb42e9c464f93",
            "type": "table",
            "value": "Table 2"
          },
          "source_id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-27",
          "id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an-scbench-2-use-evidence-4",
          "locator": {
            "document_page": 3,
            "note": null,
            "printed_page": "3",
            "source_fragment_sha256": "5af399dd9f50a252cf4fed63d58d3561db18e9cc05cdcf2622117c00f03c2929",
            "type": "table",
            "value": "Table 2"
          },
          "source_id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-27",
          "id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an-scbench-2-use-evidence-5",
          "locator": {
            "document_page": 3,
            "note": null,
            "printed_page": "3",
            "source_fragment_sha256": "b7a05555892301ba07f49c001f8755b448fccb0152d08f3f2967d2c3470c484e",
            "type": "table",
            "value": "Table 2"
          },
          "source_id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-27",
          "id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an-scbench-2-use-evidence-6",
          "locator": {
            "document_page": 3,
            "note": null,
            "printed_page": "3",
            "source_fragment_sha256": "594d2e87eaa979b255d9ff005709263a85d58c36ddd41b70bf3adfb974f269eb",
            "type": "table",
            "value": "Table 2"
          },
          "source_id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-27",
          "id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an-scbench-2-use-evidence-7",
          "locator": {
            "document_page": 3,
            "note": null,
            "printed_page": "3",
            "source_fragment_sha256": "6c258ffc743aea716119c8397fbc79fe1985a80a98455c6be301f5a399983b84",
            "type": "table",
            "value": "Table 2"
          },
          "source_id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-27",
          "id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an-scbench-2-use-evidence-8",
          "locator": {
            "document_page": 3,
            "note": null,
            "printed_page": "3",
            "source_fragment_sha256": "a119e8806f8d5cfcd3331be81d8f83a18bf5649b98c3e6885f5b131e0da71167",
            "type": "table",
            "value": "Table 2"
          },
          "source_id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-27",
          "id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an-scbench-2-use-evidence-9",
          "locator": {
            "document_page": 3,
            "note": null,
            "printed_page": "3",
            "source_fragment_sha256": "f2f00e4fbd880d94cc3c09efc848ef923bbb26d48aa638887e97310f443a0978",
            "type": "table",
            "value": "Table 2"
          },
          "source_id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-27",
          "id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an-scbench-2-use-evidence-10",
          "locator": {
            "document_page": 3,
            "note": null,
            "printed_page": "3",
            "source_fragment_sha256": "f148d29b579f618e2935016910243a85c47d22ff65aa25a6f657511d2361f635",
            "type": "table",
            "value": "Table 2"
          },
          "source_id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        }
      ],
      "id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an-scbench-2-use",
      "metric_labels": [],
      "model_ids": [
        "claude-opus-4-5",
        "claude-opus-4-6",
        "claude-sonnet-4-5",
        "gpt-5-1",
        "gpt-5-2",
        "grok-4",
        "scbench-gemini-2-5-pro-unversioned",
        "spatialbench-grok-4-1-unversioned"
      ],
      "notes": "Owner-reviewed conservative publication: the creator evaluation is retained only as a partial relationship; conflicted settings and outcomes are omitted pending manual reconciliation.",
      "relation_type": "evaluation",
      "reporting_gaps": [
        "Exact API snapshots and model release dates are not reported.",
        "Shot count and model reasoning settings are not reported.",
        "No token budget is reported.",
        "Latency confidence intervals are not reported.",
        "benchmark version",
        "realized n/scope",
        "metric",
        "numeric result",
        "prompt and tools",
        "grader and repeats"
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_reported",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-reported",
        "type": "unknown"
      },
      "status": "partial",
      "verification": {
        "last_verified": "2026-07-27",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an",
      "work_version_id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an-arxiv-v1"
    },
    {
      "benchmark_id": "single-cell-omics-arena-soar",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-08-13",
          "id": "single-cell-omics-arena-evaluation-of-large-language-m-single-cell-omics-arena-soar-1-use-evidence-1",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "ce4a54997a344e802633104f8bfc498e4a1c702af14244295aedf12660ffdc5e",
            "type": "section",
            "value": "Introduction"
          },
          "source_id": "single-cell-omics-arena-evaluation-of-large-language-m",
          "source_type": "work",
          "supports": [
            "/relation_type",
            "/benchmark_id"
          ]
        }
      ],
      "id": "single-cell-omics-arena-evaluation-of-large-language-m-single-cell-omics-arena-soar-1-use",
      "metric_labels": [],
      "model_ids": [],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "benchmark-creation",
      "reporting_gaps": [
        "No benchmark version is reported.",
        "The benchmark release date is not separately reported.",
        "The official repository license is not reported.",
        "A standalone SOAR-MultiOmics aggregate task count is not printed; PBMC and PFC counts are reported separately.",
        "The source does not describe a separately released grader artifact."
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_applicable",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-applicable",
        "type": "unknown"
      },
      "status": "non-evaluation",
      "verification": {
        "last_verified": "2026-08-13",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "single-cell-omics-arena-evaluation-of-large-language-m",
      "work_version_id": "single-cell-omics-arena-evaluation-of-large-language-m-pmc-version-1"
    },
    {
      "benchmark_id": "single-cell-omics-arena-soar",
      "benchmark_version": "initial-release",
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [
        "soar-e5d2b3e-rna-zero-shot-cot"
      ],
      "evidence": [
        {
          "accessed_date": "2026-08-13",
          "id": "soar-e5d2b3e-rna-zero-shot-cot-use-evidence",
          "locator": {
            "note": "SOAR-RNA zero-shot CoT relation, exact models, formal subset scope, metrics, and linked run.",
            "type": "repository-path",
            "value": "readme.md; soar_benchmark/configs/cell_type_annotation/experiment_soar_rna.py; soar_benchmark/task.py; soar_benchmark/datasets/soar_rna.json at commit e5d2b3e2619cb56fece5fba78fae989a67fd0c13"
          },
          "source_id": "single-cell-omics-arena-soar-official-repository-resource",
          "source_type": "resource",
          "supports": [
            "/benchmark_version",
            "/relation_type",
            "/status",
            "/model_ids",
            "/scope",
            "/metric_labels",
            "/evaluation_run_ids",
            "/reporting_gaps",
            "/notes"
          ]
        }
      ],
      "id": "soar-e5d2b3e-rna-zero-shot-cot-use",
      "metric_labels": [
        "R-1",
        "R-2",
        "R-L",
        "MET.",
        "B-1",
        "B-2",
        "BLEU"
      ],
      "model_ids": [
        "openai-gpt-4o-2024-05-13",
        "openai-gpt-4o-mini-2024-07-18"
      ],
      "notes": "Official commit-pinned SOAR-RNA two-call zero-shot chain-of-thought result snapshot; the run covers the formal SOAR-RNA subset, not the full SOAR suite.",
      "relation_type": "evaluation",
      "reporting_gaps": [
        "repeat count",
        "confidence intervals",
        "contamination or decontamination analysis"
      ],
      "scope": {
        "n": 1191,
        "reporting_status": "reported",
        "selection": "formal-subset",
        "selection_method": "All 1,191 records in the pinned SOAR-RNA JSON artifact, iterated in repository order with shuffle disabled.",
        "subset_id": "soar-rna",
        "subset_kind": "formal-subset",
        "type": "subset"
      },
      "status": "normalized",
      "verification": {
        "last_verified": "2026-08-13",
        "notes": "Normalized relationship is limited to independently verified high-confidence claims.",
        "status": "verified"
      },
      "work_id": "single-cell-omics-arena-soar-repository-result-snapshot",
      "work_version_id": "single-cell-omics-arena-soar-repository-result-snapshot-e5d2b3e"
    },
    {
      "benchmark_id": "single-cell-omics-arena-soar",
      "benchmark_version": "initial-release",
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [
        "soar-e5d2b3e-rna-zero-shot"
      ],
      "evidence": [
        {
          "accessed_date": "2026-08-13",
          "id": "soar-e5d2b3e-rna-zero-shot-use-evidence",
          "locator": {
            "note": "SOAR-RNA zero-shot relation, exact models, formal subset scope, metrics, and linked run.",
            "type": "repository-path",
            "value": "readme.md; soar_benchmark/configs/cell_type_annotation/experiment_soar_rna.py; soar_benchmark/datasets/soar_rna.json at commit e5d2b3e2619cb56fece5fba78fae989a67fd0c13"
          },
          "source_id": "single-cell-omics-arena-soar-official-repository-resource",
          "source_type": "resource",
          "supports": [
            "/benchmark_version",
            "/relation_type",
            "/status",
            "/model_ids",
            "/scope",
            "/metric_labels",
            "/evaluation_run_ids",
            "/reporting_gaps",
            "/notes"
          ]
        }
      ],
      "id": "soar-e5d2b3e-rna-zero-shot-use",
      "metric_labels": [
        "R-1",
        "R-2",
        "R-L",
        "MET.",
        "B-1",
        "B-2",
        "BLEU"
      ],
      "model_ids": [
        "openai-gpt-4o-2024-05-13",
        "openai-gpt-4o-mini-2024-07-18"
      ],
      "notes": "Official commit-pinned SOAR-RNA zero-shot result snapshot; the run covers the formal SOAR-RNA subset, not the full SOAR suite.",
      "relation_type": "evaluation",
      "reporting_gaps": [
        "repeat count",
        "confidence intervals",
        "contamination or decontamination analysis"
      ],
      "scope": {
        "n": 1191,
        "reporting_status": "reported",
        "selection": "formal-subset",
        "selection_method": "All 1,191 records in the pinned SOAR-RNA JSON artifact, iterated in repository order with shuffle disabled.",
        "subset_id": "soar-rna",
        "subset_kind": "formal-subset",
        "type": "subset"
      },
      "status": "normalized",
      "verification": {
        "last_verified": "2026-08-13",
        "notes": "Normalized relationship is limited to independently verified high-confidence claims.",
        "status": "verified"
      },
      "work_id": "single-cell-omics-arena-soar-repository-result-snapshot",
      "work_version_id": "single-cell-omics-arena-soar-repository-result-snapshot-e5d2b3e"
    },
    {
      "benchmark_id": "spatialbench",
      "benchmark_version": "paper-v2",
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "spatialbench-preprint-creation-use-evidence",
          "locator": {
            "note": "Introduces and constructs SpatialBench.",
            "type": "page",
            "value": "arXiv v2 abstract and §§2.1, 3.1–3.4"
          },
          "source_id": "spatialbench-preprint",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/relation_type",
            "/status",
            "/scope",
            "/notes"
          ]
        }
      ],
      "id": "spatialbench-preprint-creation",
      "metric_labels": [],
      "model_ids": [],
      "notes": "Creator preprint defining the 146-problem paper-v2 benchmark snapshot.",
      "relation_type": "benchmark-creation",
      "reporting_gaps": [],
      "scope": {
        "n": null,
        "reporting_status": "not_applicable",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-applicable",
        "type": "unknown"
      },
      "status": "non-evaluation",
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Benchmark-creation relation is separate from the paper's model evaluations.",
        "status": "verified"
      },
      "work_id": "spatialbench-preprint",
      "work_version_id": "spatialbench-preprint-v2"
    },
    {
      "benchmark_id": "spatialbench",
      "benchmark_version": "paper-v2",
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [
        "spatialbench-paper-v2-base",
        "spatialbench-paper-v2-claude-code",
        "spatialbench-paper-v2-latch"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "spatialbench-preprint-evaluation-use-evidence",
          "locator": {
            "note": "Model and harness results over all 146 evaluations.",
            "type": "table",
            "value": "arXiv v2 Tables 1 and 4; §§2.2, 2.5, 3.5–3.7"
          },
          "source_id": "spatialbench-preprint",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/relation_type",
            "/status",
            "/model_ids",
            "/scope",
            "/metric_labels",
            "/evaluation_run_ids",
            "/notes"
          ]
        }
      ],
      "id": "spatialbench-preprint-evaluation",
      "metric_labels": [
        "Accuracy",
        "Steps",
        "Latency",
        "Cost"
      ],
      "model_ids": [
        "claude-opus-4-5",
        "claude-sonnet-4-5",
        "gpt-5-2",
        "gpt-5-1",
        "gemini-2-5-pro",
        "grok-4",
        "spatialbench-grok-4-1-unversioned"
      ],
      "notes": "Base, Claude Code, and Latch harnesses are normalized as separate runs and comparability groups.",
      "relation_type": "evaluation",
      "reporting_gaps": [],
      "scope": {
        "n": 146,
        "reporting_status": "reported",
        "selection": null,
        "selection_method": "all 146 paper-v2 evaluations",
        "subset_id": null,
        "subset_kind": "not-applicable",
        "type": "full"
      },
      "status": "normalized",
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Full paper-v2 evaluation relation backed by labeled tables and versioned methods.",
        "status": "verified"
      },
      "work_id": "spatialbench-preprint",
      "work_version_id": "spatialbench-preprint-v2"
    },
    {
      "benchmark_id": "spatialbench",
      "benchmark_version": "repo-159-5042c4f",
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "spatialbench-repository-creation-use-evidence",
          "locator": {
            "note": "Defines the revised 159-evaluation version.",
            "type": "repository-path",
            "value": "README.md and CHANGELOG.md at commit 5042c4f3ee597da1590650c7b894d068ae968e26"
          },
          "source_id": "spatialbench-repository-resource",
          "source_type": "resource",
          "supports": [
            "/benchmark_version",
            "/relation_type",
            "/status",
            "/scope",
            "/notes"
          ]
        }
      ],
      "id": "spatialbench-repository-creation",
      "metric_labels": [],
      "model_ids": [],
      "notes": "Creator-maintained revised 159-evaluation benchmark snapshot fixed to the registered commit.",
      "relation_type": "benchmark-creation",
      "reporting_gaps": [],
      "scope": {
        "n": null,
        "reporting_status": "not_applicable",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-applicable",
        "type": "unknown"
      },
      "status": "non-evaluation",
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Revised benchmark release relation is separate from its current result table.",
        "status": "verified"
      },
      "work_id": "spatialbench-repository-release",
      "work_version_id": "spatialbench-repository-release-2026-06-10"
    },
    {
      "benchmark_id": "spatialbench",
      "benchmark_version": "repo-159-5042c4f",
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [
        "spatialbench-repo-159-mini-swe-agent",
        "spatialbench-repo-159-claude-code",
        "spatialbench-repo-159-pi",
        "spatialbench-repo-159-openai-codex"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "spatialbench-repository-evaluation-use-evidence",
          "locator": {
            "note": "Full 159-evaluation methods and harness-specific result rows.",
            "type": "repository-path",
            "value": "METHODS.md and results/model_results.csv at commit 5042c4f3ee597da1590650c7b894d068ae968e26"
          },
          "source_id": "spatialbench-repository-resource",
          "source_type": "resource",
          "supports": [
            "/benchmark_version",
            "/relation_type",
            "/status",
            "/model_ids",
            "/scope",
            "/metric_labels",
            "/evaluation_run_ids",
            "/notes"
          ]
        }
      ],
      "id": "spatialbench-repository-evaluation",
      "metric_labels": [
        "Accuracy",
        "Cost",
        "Duration"
      ],
      "model_ids": [
        "gpt-5-5",
        "gpt-5-4",
        "gemini-3-5-flash",
        "claude-opus-4-8",
        "claude-opus-4-6",
        "claude-opus-4-7",
        "gemini-3-1-pro-preview",
        "gpt-5-2",
        "grok-4-20-beta-0309-reasoning",
        "claude-sonnet-4-6",
        "claude-opus-4-5",
        "claude-sonnet-4-5",
        "gpt-5-1",
        "grok-4-1-fast-reasoning",
        "grok-4",
        "gemini-2-5-pro"
      ],
      "notes": "mini-swe-agent, Claude Code, Pi, and OpenAI Codex are separate normalized runs and comparability groups.",
      "relation_type": "evaluation",
      "reporting_gaps": [],
      "scope": {
        "n": 159,
        "reporting_status": "reported",
        "selection": null,
        "selection_method": "all 159 evaluations",
        "subset_id": null,
        "subset_kind": "not-applicable",
        "type": "full"
      },
      "status": "normalized",
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Current full-suite results and protocol are tied to the registered repository commit.",
        "status": "verified"
      },
      "work_id": "spatialbench-repository-release",
      "work_version_id": "spatialbench-repository-release-2026-06-10"
    },
    {
      "benchmark_id": "biomysterybench",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-biomysterybench-1-use-evidence-1",
          "locator": {
            "document_page": 188,
            "note": null,
            "printed_page": "188",
            "source_fragment_sha256": "41c0e2e79bc42390b4120af5d2f976e12371fb5fc35a403976b423575ecb487b",
            "type": "section",
            "value": "Section 8.17.1 BioMysteryBench"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-biomysterybench-1-use-evidence-2",
          "locator": {
            "document_page": 188,
            "note": null,
            "printed_page": "188",
            "source_fragment_sha256": "cc252a6a42efdbb7acb08b7fa5fe2ee99854076fe89da66a4ed4af796d1db3b0",
            "type": "section",
            "value": "Section 8.17.1"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-biomysterybench-1-use-evidence-3",
          "locator": {
            "document_page": 188,
            "note": null,
            "printed_page": "188",
            "source_fragment_sha256": "ca915a08b6a6edeff979413dde411c4b63967292650907ff20ffbc635b9c85dc",
            "type": "section",
            "value": "Section 8.17.1 BioMysteryBench"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-biomysterybench-1-use-evidence-4",
          "locator": {
            "document_page": 188,
            "note": null,
            "printed_page": "188",
            "source_fragment_sha256": "e2c466511eaa765f1c6fd9f70b9e1a41659e9134c5bb669b369b36cfcb926633",
            "type": "section",
            "value": "Section 8.17.1 BioMysteryBench"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-biomysterybench-1-use-evidence-5",
          "locator": {
            "document_page": 188,
            "note": null,
            "printed_page": "188",
            "source_fragment_sha256": "b761b2856b4bece9fcf8278a919cb0852df2681454f0499b01954ef3e1685822",
            "type": "section",
            "value": "Section 8.17.1 BioMysteryBench"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-biomysterybench-1-use-evidence-6",
          "locator": {
            "document_page": 188,
            "note": null,
            "printed_page": "188",
            "source_fragment_sha256": "e2c466511eaa765f1c6fd9f70b9e1a41659e9134c5bb669b369b36cfcb926633",
            "type": "section",
            "value": "Section 8.17.1 BioMysteryBench"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-biomysterybench-1-use-evidence-7",
          "locator": {
            "document_page": 188,
            "note": null,
            "printed_page": "188",
            "source_fragment_sha256": "b761b2856b4bece9fcf8278a919cb0852df2681454f0499b01954ef3e1685822",
            "type": "section",
            "value": "Section 8.17.1 BioMysteryBench"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-biomysterybench-1-use-evidence-8",
          "locator": {
            "document_page": 190,
            "note": null,
            "printed_page": "190",
            "source_fragment_sha256": "22c5685582438b914c9166e2b72f11a85688cd78bb0a751c965552c2caca8b93",
            "type": "figure",
            "value": "Figure 8.17.6.A, BioMysteryBench panel"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-biomysterybench-1-use-evidence-9",
          "locator": {
            "document_page": 188,
            "note": null,
            "printed_page": "188",
            "source_fragment_sha256": "590a72314caf54f70dae968253ecb61f1d479aeb738423c805000e075b95d315",
            "type": "section",
            "value": "Section 8.17.1 BioMysteryBench"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-biomysterybench-1-use-evidence-10",
          "locator": {
            "document_page": 188,
            "note": null,
            "printed_page": "188",
            "source_fragment_sha256": "d545a1f43c6d993e90df9b8fbf30b3cc9347186220147da43202442677a4da30",
            "type": "section",
            "value": "Section 8.17.1 BioMysteryBench"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-biomysterybench-1-use-evidence-11",
          "locator": {
            "document_page": 188,
            "note": null,
            "printed_page": "188",
            "source_fragment_sha256": "2c297991a44f1e9a905b155136ff8f98bd0879390aba2a52f42edfcf6da27ab5",
            "type": "section",
            "value": "Section 8.17.1 BioMysteryBench"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-biomysterybench-1-use-evidence-12",
          "locator": {
            "document_page": 188,
            "note": null,
            "printed_page": "188",
            "source_fragment_sha256": "ef386d95d3b276759763a23acd841eac8e27c0f21b53f9ba6bc69cc7fcd1a5b2",
            "type": "section",
            "value": "Section 8.17.1 BioMysteryBench"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        }
      ],
      "id": "system-card-claude-opus-5-biomysterybench-1-use",
      "metric_labels": [
        "Score"
      ],
      "model_ids": [
        "anthropic-claude-mythos-5",
        "anthropic-claude-opus-5",
        "anthropic-claude-sonnet-5",
        "claude-opus-4-8"
      ],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "evaluation",
      "reporting_gaps": [
        "Exact benchmark version is not reported.",
        "Overall total and current Human Solvable and Human Difficult subset sizes are not reported; only removal counts are given.",
        "Prompt, shots, reasoning settings, budget, seed, repeats, grader, and human review are not reported.",
        "Score definition and aggregation are not reported.",
        "benchmark version",
        "realized n/scope"
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_reported",
        "selection": null,
        "selection_method": "Human Difficult: problems unsolved by humans with an objective ground-truth solution",
        "subset_id": "human-difficult",
        "subset_kind": "formal-subset",
        "type": "subset"
      },
      "status": "partial",
      "verification": {
        "last_verified": "2026-07-29",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "system-card-claude-opus-5",
      "work_version_id": "system-card-claude-opus-5-2026-07-24"
    },
    {
      "benchmark_id": "proteingym",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-proteingym-4-use-evidence-1",
          "locator": {
            "document_page": 189,
            "note": null,
            "printed_page": "189",
            "source_fragment_sha256": "502ee6c6227dcdaa69e4518f8b5d97cb5b3ab3ab72525a083e9cdb6f690c7db5",
            "type": "section",
            "value": "Section 8.17.3 ProteinGym Hard"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-proteingym-4-use-evidence-2",
          "locator": {
            "document_page": 189,
            "note": null,
            "printed_page": "189",
            "source_fragment_sha256": "20bfa13469fb999cbb53d1cc2e2969aad494b769a422e5799626844b94ada70f",
            "type": "section",
            "value": "Section 8.17.3 ProteinGym Hard"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-proteingym-4-use-evidence-3",
          "locator": {
            "document_page": 189,
            "note": null,
            "printed_page": "189",
            "source_fragment_sha256": "404298efcd4a0832ecbe6f08088daa6ee7287ab827944028d979216d02bd30f3",
            "type": "section",
            "value": "Section 8.17.3 ProteinGym Hard"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-proteingym-4-use-evidence-4",
          "locator": {
            "document_page": 189,
            "note": null,
            "printed_page": "189",
            "source_fragment_sha256": "20bfa13469fb999cbb53d1cc2e2969aad494b769a422e5799626844b94ada70f",
            "type": "section",
            "value": "Section 8.17.3 ProteinGym Hard"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-proteingym-4-use-evidence-5",
          "locator": {
            "document_page": 189,
            "note": null,
            "printed_page": "189",
            "source_fragment_sha256": "404298efcd4a0832ecbe6f08088daa6ee7287ab827944028d979216d02bd30f3",
            "type": "section",
            "value": "Section 8.17.3 ProteinGym Hard"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-proteingym-4-use-evidence-6",
          "locator": {
            "document_page": 189,
            "note": null,
            "printed_page": "189",
            "source_fragment_sha256": "b04f06cda1c343180660a02687918e53030accb598ef434b2173cdc90cb7f2d4",
            "type": "section",
            "value": "Section 8.17.3 ProteinGym Hard"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-proteingym-4-use-evidence-7",
          "locator": {
            "document_page": 189,
            "note": null,
            "printed_page": "189",
            "source_fragment_sha256": "57b74c83bdc15f21b3ff51d86cf0090371ec1adae398623cf63d81229993d86b",
            "type": "section",
            "value": "Section 8.17.3 ProteinGym Hard"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-proteingym-4-use-evidence-8",
          "locator": {
            "document_page": 189,
            "note": null,
            "printed_page": "189",
            "source_fragment_sha256": "01aceff176cfe23d5165f5ebb108a37ba6db38a48294262efb39c294553b4929",
            "type": "section",
            "value": "Section 8.17.3 ProteinGym Hard"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-proteingym-4-use-evidence-9",
          "locator": {
            "document_page": 189,
            "note": null,
            "printed_page": "189",
            "source_fragment_sha256": "d2e8db20a168c034a1c35a2f208502a117b0a8b552e09411c53788d8a4c714ac",
            "type": "section",
            "value": "Section 8.17.3 ProteinGym Hard"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-proteingym-4-use-evidence-10",
          "locator": {
            "document_page": 189,
            "note": null,
            "printed_page": "189",
            "source_fragment_sha256": "b1aeedc45f1822c2afe404cf07a45e02cf58ae788a230bc1ff2c3eb049d93b26",
            "type": "section",
            "value": "Section 8.17.3 ProteinGym Hard"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        }
      ],
      "id": "system-card-claude-opus-5-proteingym-4-use",
      "metric_labels": [
        "rank correlation against real lab measurements"
      ],
      "model_ids": [
        "anthropic-claude-mythos-5",
        "anthropic-claude-opus-5",
        "anthropic-claude-sonnet-5",
        "claude-opus-4-8"
      ],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "evaluation",
      "reporting_gaps": [
        "Exact ProteinGym version is not reported.",
        "The Hard subset size and selection criteria are not reported.",
        "Prompt, shots, reasoning settings, budget, seed, repeats, grader, and human review are not reported.",
        "The rank-correlation variant and aggregation procedure are not reported.",
        "benchmark version",
        "realized n/scope"
      ],
      "scope": {
        "n": null,
        "reporting_status": "not_reported",
        "selection": "filtered",
        "selection_method": "Subset of mutant protein sequences ranked against the wild type sequence",
        "subset_id": "hard",
        "subset_kind": "not-reported",
        "type": "unknown"
      },
      "status": "partial",
      "verification": {
        "last_verified": "2026-07-29",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "system-card-claude-opus-5",
      "work_version_id": "system-card-claude-opus-5-2026-07-24"
    },
    {
      "benchmark_id": "scbench",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-scbench-3-use-evidence-1",
          "locator": {
            "document_page": 188,
            "note": null,
            "printed_page": "188",
            "source_fragment_sha256": "d7e5b637c8345c5cd158fe0289396b8a4b3fd717bf51a83228660f8a0e41729c",
            "type": "section",
            "value": "Section 8.17.2 LatchBio Bioinformatics"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-scbench-3-use-evidence-2",
          "locator": {
            "document_page": 188,
            "note": null,
            "printed_page": "188",
            "source_fragment_sha256": "e36a3723aca33ffca31c23f9345fcda7afc441528be390fbb42ab0c78bd4f754",
            "type": "section",
            "value": "Section 8.17.2 LatchBio Bioinformatics"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-scbench-3-use-evidence-3",
          "locator": {
            "document_page": 188,
            "note": null,
            "printed_page": "188",
            "source_fragment_sha256": "5125be62618415e164e9d7f6a234ec29ece96497d0c5d1635e5558ac682d51e4",
            "type": "section",
            "value": "Section 8.17.2 LatchBio Bioinformatics"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-scbench-3-use-evidence-4",
          "locator": {
            "document_page": 188,
            "note": null,
            "printed_page": "188",
            "source_fragment_sha256": "777affffdc84bc54744f9cfb2acbe11edd5e2af1a140131f4b8ecca830351f2f",
            "type": "section",
            "value": "Section 8.17.2 LatchBio Bioinformatics"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-scbench-3-use-evidence-5",
          "locator": {
            "document_page": 190,
            "note": null,
            "printed_page": "190",
            "source_fragment_sha256": "22c5685582438b914c9166e2b72f11a85688cd78bb0a751c965552c2caca8b93",
            "type": "figure",
            "value": "Figure 8.17.6.A, LatchBio Bioinformatics panel"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-scbench-3-use-evidence-6",
          "locator": {
            "document_page": 188,
            "note": null,
            "printed_page": "188",
            "source_fragment_sha256": "448232ef2adc1aeb1097770f7201e36df5ce86a07ed2333ca261665ec2b48002",
            "type": "section",
            "value": "Section 8.17.2, SingleCellBench"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-scbench-3-use-evidence-7",
          "locator": {
            "document_page": 188,
            "note": null,
            "printed_page": "188",
            "source_fragment_sha256": "467684e37d7de337a7582c274a1c8fc5d3e23ac4a2bc8fd8df76e072d714283c",
            "type": "section",
            "value": "Section 8.17.2, SingleCellBench"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-scbench-3-use-evidence-8",
          "locator": {
            "document_page": 188,
            "note": null,
            "printed_page": "188",
            "source_fragment_sha256": "50a6fb5ac5f6e5b2321e2856df46286c01a4caf2f1c1536f79dbf33c0755f6f3",
            "type": "section",
            "value": "Section 8.17.2, SingleCellBench"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-scbench-3-use-evidence-9",
          "locator": {
            "document_page": 188,
            "note": null,
            "printed_page": "188",
            "source_fragment_sha256": "f86833ee25cfd235ae251315f49d7783de851613e17c992b4d72f77419b4fa99",
            "type": "section",
            "value": "Section 8.17.2, SingleCellBench"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        }
      ],
      "id": "system-card-claude-opus-5-scbench-3-use",
      "metric_labels": [
        "Score"
      ],
      "model_ids": [
        "anthropic-claude-mythos-5",
        "anthropic-claude-opus-5",
        "anthropic-claude-sonnet-5",
        "claude-opus-4-8"
      ],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "evaluation",
      "reporting_gaps": [
        "Exact benchmark version is not reported.",
        "Per-workflow problem counts are not reported.",
        "Prompt, shots, reasoning settings, budget, seed, repeats, grader, and human review are not reported.",
        "Score definition and aggregation are not reported.",
        "benchmark version"
      ],
      "scope": {
        "n": 195,
        "reporting_status": "not_reported",
        "selection": null,
        "selection_method": null,
        "subset_id": null,
        "subset_kind": "not-reported",
        "type": "unknown"
      },
      "status": "partial",
      "verification": {
        "last_verified": "2026-07-29",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "system-card-claude-opus-5",
      "work_version_id": "system-card-claude-opus-5-2026-07-24"
    },
    {
      "benchmark_id": "spatialbench",
      "benchmark_version": null,
      "entity_type": "benchmark_use",
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-spatialbench-2-use-evidence-1",
          "locator": {
            "document_page": 188,
            "note": null,
            "printed_page": "188",
            "source_fragment_sha256": "23d744fa62eb1e7a4836d687d2fb1670ea7adcd269bbb2f558fd4f1060e8d2de",
            "type": "section",
            "value": "Section 8.17.2 LatchBio Bioinformatics"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/relation_type"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-spatialbench-2-use-evidence-2",
          "locator": {
            "document_page": 188,
            "note": null,
            "printed_page": "188",
            "source_fragment_sha256": "a74f915e74a18a6d24c9d54fef9faa0c52cfa701df3ac07b6edd06e4f0923ff1",
            "type": "section",
            "value": "Section 8.17.2 LatchBio Bioinformatics"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/benchmark_id"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-spatialbench-2-use-evidence-3",
          "locator": {
            "document_page": 188,
            "note": null,
            "printed_page": "188",
            "source_fragment_sha256": "ba27967f0f97fe775985df69bdd145d0199a75c0b29e2d8db13feae41c5835ed",
            "type": "section",
            "value": "Section 8.17.2 LatchBio Bioinformatics"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-spatialbench-2-use-evidence-4",
          "locator": {
            "document_page": 188,
            "note": null,
            "printed_page": "188",
            "source_fragment_sha256": "a74f915e74a18a6d24c9d54fef9faa0c52cfa701df3ac07b6edd06e4f0923ff1",
            "type": "section",
            "value": "Section 8.17.2 LatchBio Bioinformatics"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-spatialbench-2-use-evidence-5",
          "locator": {
            "document_page": 188,
            "note": null,
            "printed_page": "188",
            "source_fragment_sha256": "6bd88a5fd6f61aa463e6639e168dd3edffaa001a29ca47a6319b4f436758d9f9",
            "type": "section",
            "value": "Section 8.17.2 LatchBio Bioinformatics"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-spatialbench-2-use-evidence-6",
          "locator": {
            "document_page": 188,
            "note": null,
            "printed_page": "188",
            "source_fragment_sha256": "6bd88a5fd6f61aa463e6639e168dd3edffaa001a29ca47a6319b4f436758d9f9",
            "type": "section",
            "value": "Section 8.17.2 LatchBio Bioinformatics"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-spatialbench-2-use-evidence-7",
          "locator": {
            "document_page": 188,
            "note": null,
            "printed_page": "188",
            "source_fragment_sha256": "23d744fa62eb1e7a4836d687d2fb1670ea7adcd269bbb2f558fd4f1060e8d2de",
            "type": "section",
            "value": "Section 8.17.2 LatchBio Bioinformatics"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/metric_labels"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-spatialbench-2-use-evidence-8",
          "locator": {
            "document_page": 188,
            "note": null,
            "printed_page": "188",
            "source_fragment_sha256": "ddcfac29537e1d474f177ff2a490642ea789958433eb19b967526f3bbbc82cdf",
            "type": "section",
            "value": "Section 8.17.2, SpatialBench Verified"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-spatialbench-2-use-evidence-9",
          "locator": {
            "document_page": 188,
            "note": null,
            "printed_page": "188",
            "source_fragment_sha256": "b439385064af327cf5ba04c5d0d679fad803a89a9a4f614e0f8e6562bead6161",
            "type": "section",
            "value": "Section 8.17.2, SpatialBench Verified"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-spatialbench-2-use-evidence-10",
          "locator": {
            "document_page": 188,
            "note": null,
            "printed_page": "188",
            "source_fragment_sha256": "eb346c26a57c599b8ac9c8d41f922918ab839a19177fea61e7700871da666d27",
            "type": "section",
            "value": "Section 8.17.2, SpatialBench Verified"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "system-card-claude-opus-5-spatialbench-2-use-evidence-11",
          "locator": {
            "document_page": 188,
            "note": null,
            "printed_page": "188",
            "source_fragment_sha256": "0da462e0ba43830bc8537490b3a332669c88c2dd9fce5ec0d3eb268e0990d8c6",
            "type": "section",
            "value": "Section 8.17.2, SpatialBench Verified"
          },
          "source_id": "system-card-claude-opus-5",
          "source_type": "work",
          "supports": [
            "/model_ids"
          ]
        }
      ],
      "id": "system-card-claude-opus-5-spatialbench-2-use",
      "metric_labels": [
        "Score"
      ],
      "model_ids": [
        "anthropic-claude-mythos-5",
        "anthropic-claude-opus-5",
        "anthropic-claude-sonnet-5",
        "claude-opus-4-8"
      ],
      "notes": "AI-assisted double-pass extraction; values are limited to independently supported claims.",
      "relation_type": "evaluation",
      "reporting_gaps": [
        "The source does not identify an official benchmark version or artifact revision for the Verified qualifier.",
        "Prompt, shots, reasoning settings, budget, seed, repeats, grader, and human review are not reported.",
        "Score definition and aggregation are not reported.",
        "benchmark version"
      ],
      "scope": {
        "n": 115,
        "reporting_status": "not_reported",
        "selection": null,
        "selection_method": "Externally validated problems",
        "subset_id": null,
        "subset_kind": "not-reported",
        "type": "unknown"
      },
      "status": "partial",
      "verification": {
        "last_verified": "2026-07-29",
        "notes": "Ready for production only after owner approval of this paper-intake PR's exact head SHA.",
        "status": "verified"
      },
      "work_id": "system-card-claude-opus-5",
      "work_version_id": "system-card-claude-opus-5-2026-07-24"
    }
  ],
  "benchmarks": [
    {
      "access": {
        "artifacts": "Public repository containing experimental data, computational data, parent structures, and documentation.",
        "biosafety_notes": null,
        "grader": "Not established by the independently verified metadata claims.",
        "level": "fully-open",
        "license": null,
        "tasks": "Experimental binding free-energy changes and related PDB identifiers are provided."
      },
      "aliases": [
        "Antibody-Bind"
      ],
      "audit": {
        "audited_date": "2026-08-14",
        "notes": "Automated double-pass extraction plus deterministic official-resource pin. Owner review preserved the corroborated root total and excluded conflicted appendix inventory subcounts.",
        "status": "audited-with-caveats",
        "unresolved_fields": 2
      },
      "benchmark_use_ids": [
        "ab-bind-antibody-binding-mutational-database-for-compu-ab-bind-1-use",
        "ab-bind-antibody-binding-mutational-database-for-compu-ab-bind-2-use",
        "ab-bind-antibody-binding-mutational-database-for-compu-ab-bind-3-use",
        "explainable-protein-protein-binding-affinity-predictio-ab-bind-5-use",
        "explainable-protein-protein-binding-affinity-predictio-ab-bind-6-use",
        "explainable-protein-protein-binding-affinity-predictio-ab-bind-7-use",
        "explainable-protein-protein-binding-affinity-predictio-ab-bind-8-use"
      ],
      "capabilities": [
        "prediction",
        "classification",
        "regression",
        "optimization"
      ],
      "coverage_notes": [],
      "domains": [
        "protein-protein-binding",
        "antibody-antigen"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "ab-bind-antibody-binding-mutational-database-for-compu",
        "explainable-protein-protein-binding-affinity-predictio"
      ],
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-08-14",
          "id": "ab-bind-automated-metadata-1-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "02b14c00e15db1358793aaa85439c19a5488bb3501a92899a3d93bf62edc36ec",
            "type": "repository-path",
            "value": "sarahsirin/AB-Bind-Database@f5af13df80000ad9e438a6ccf1147bd5f9d55dba > file_paths"
          },
          "source_id": "ab-bind-antibody-binding-mutational-database-for-compu",
          "source_type": "work",
          "supports": [
            "/access/artifacts"
          ]
        },
        {
          "accessed_date": "2026-08-14",
          "id": "ab-bind-automated-metadata-2-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "d38dfce4a26f6ef9d390ede33f2ab9c49cf39bbbfbaf7cac2b6dd78a59539929",
            "type": "repository-path",
            "value": "sarahsirin/AB-Bind-Database@f5af13df80000ad9e438a6ccf1147bd5f9d55dba"
          },
          "source_id": "ab-bind-antibody-binding-mutational-database-for-compu",
          "source_type": "work",
          "supports": [
            "/access/level"
          ]
        },
        {
          "accessed_date": "2026-08-14",
          "id": "ab-bind-automated-metadata-3-evidence",
          "locator": {
            "document_page": 12,
            "note": null,
            "printed_page": "404",
            "source_fragment_sha256": "2f3c5171b0f6a4652b03644476bf4366e81b2447e1fb3e18861b80f34ad7dcb9",
            "type": "section",
            "value": "Materials and Methods > Datasets"
          },
          "source_id": "ab-bind-antibody-binding-mutational-database-for-compu",
          "source_type": "work",
          "supports": [
            "/access/tasks"
          ]
        },
        {
          "accessed_date": "2026-08-14",
          "id": "ab-bind-automated-metadata-4-evidence",
          "locator": {
            "document_page": 1,
            "note": null,
            "printed_page": "393",
            "source_fragment_sha256": "5b9f01e36745fe232f491417c34fff66a53682268df5bbba73b580b0fbdac323",
            "type": "section",
            "value": "Abstract"
          },
          "source_id": "ab-bind-antibody-binding-mutational-database-for-compu",
          "source_type": "work",
          "supports": [
            "/aliases"
          ]
        },
        {
          "accessed_date": "2026-08-14",
          "id": "ab-bind-automated-metadata-5-evidence",
          "locator": {
            "document_page": 1,
            "note": null,
            "printed_page": "393",
            "source_fragment_sha256": "c3f04cf1bf4b9ee1946254b4b8ef8f8d63a76c450240738703eb412bd6d39300",
            "type": "section",
            "value": "Abstract and keywords"
          },
          "source_id": "ab-bind-antibody-binding-mutational-database-for-compu",
          "source_type": "work",
          "supports": [
            "/capabilities"
          ]
        },
        {
          "accessed_date": "2026-08-14",
          "id": "ab-bind-automated-metadata-6-evidence",
          "locator": {
            "document_page": 3,
            "note": null,
            "printed_page": "395",
            "source_fragment_sha256": "063bb4888e5ffb17a428a7a53468f54be6c4ce96af6b6f36b7810ac041f52b1c",
            "type": "section",
            "value": "Introduction"
          },
          "source_id": "ab-bind-antibody-binding-mutational-database-for-compu",
          "source_type": "work",
          "supports": [
            "/domains"
          ]
        },
        {
          "accessed_date": "2026-08-14",
          "id": "ab-bind-automated-metadata-7-evidence",
          "locator": {
            "document_page": 12,
            "note": null,
            "printed_page": "404",
            "source_fragment_sha256": "acc944e8fcb9189a3b61b114d4e606a978006ab143a62c97f0ed09f377662c0d",
            "type": "section",
            "value": "Discussion > Summary"
          },
          "source_id": "ab-bind-antibody-binding-mutational-database-for-compu",
          "source_type": "work",
          "supports": [
            "/kind"
          ]
        },
        {
          "accessed_date": "2026-08-14",
          "id": "ab-bind-automated-metadata-8-evidence",
          "locator": {
            "document_page": 12,
            "note": null,
            "printed_page": "404",
            "source_fragment_sha256": "2f3c5171b0f6a4652b03644476bf4366e81b2447e1fb3e18861b80f34ad7dcb9",
            "type": "section",
            "value": "Materials and Methods > Datasets"
          },
          "source_id": "ab-bind-antibody-binding-mutational-database-for-compu",
          "source_type": "work",
          "supports": [
            "/modalities"
          ]
        },
        {
          "accessed_date": "2026-08-14",
          "id": "ab-bind-automated-metadata-9-evidence",
          "locator": {
            "document_page": 1,
            "note": null,
            "printed_page": "393",
            "source_fragment_sha256": "5b9f01e36745fe232f491417c34fff66a53682268df5bbba73b580b0fbdac323",
            "type": "section",
            "value": "Abstract"
          },
          "source_id": "ab-bind-antibody-binding-mutational-database-for-compu",
          "source_type": "work",
          "supports": [
            "/name"
          ]
        },
        {
          "accessed_date": "2026-08-14",
          "id": "ab-bind-automated-metadata-10-evidence",
          "locator": {
            "document_page": 1,
            "note": null,
            "printed_page": "393",
            "source_fragment_sha256": "7c2d2b3b86c63b8fe103171062929b9044737cba65336f95f33aa693d7fa9785",
            "type": "page",
            "value": "Author list and affiliations"
          },
          "source_id": "ab-bind-antibody-binding-mutational-database-for-compu",
          "source_type": "work",
          "supports": [
            "/organizations"
          ]
        },
        {
          "accessed_date": "2026-08-14",
          "id": "ab-bind-automated-metadata-11-evidence",
          "locator": {
            "document_page": 1,
            "note": null,
            "printed_page": "393",
            "source_fragment_sha256": "1a2ec1d8f87234dca45061bef2dc6cc298c6fd62e256fb71842e5970b1dae5bb",
            "type": "section",
            "value": "Abstract"
          },
          "source_id": "ab-bind-antibody-binding-mutational-database-for-compu",
          "source_type": "work",
          "supports": [
            "/summary"
          ]
        },
        {
          "accessed_date": "2026-08-14",
          "id": "ab-bind-automated-metadata-12-evidence",
          "locator": {
            "document_page": 1,
            "note": null,
            "printed_page": "393",
            "source_fragment_sha256": "b1980faf945719c862c810a4aa93417ccb026e611b4062b19b9cde16ce17f7ba",
            "type": "section",
            "value": "Abstract"
          },
          "source_id": "ab-bind-antibody-binding-mutational-database-for-compu",
          "source_type": "work",
          "supports": [
            "/task_formats"
          ]
        },
        {
          "accessed_date": "2026-08-14",
          "id": "ab-bind-automated-bibliographic-evidence",
          "locator": {
            "document_page": null,
            "note": "Resolved from the canonical paper identifier during intake.",
            "printed_page": null,
            "source_fragment_sha256": "b7f8cdd66f73a2bb6203666680363607c0d0e6feb0485a208be90c1c297e3572",
            "type": "other",
            "value": "Crossref bibliographic metadata"
          },
          "source_id": "ab-bind-antibody-binding-mutational-database-for-compu",
          "source_type": "work",
          "supports": [
            "/release_date"
          ]
        },
        {
          "accessed_date": "2026-08-14",
          "id": "ab-bind-automated-count-evidence",
          "locator": {
            "document_page": 3,
            "note": null,
            "printed_page": "395",
            "source_fragment_sha256": "d4faef71f766ca870813d135356c73eb29d3bdb2affab9aa34502784dd9f4cee",
            "type": "section",
            "value": "Results > AB-Bind database"
          },
          "source_id": "ab-bind-antibody-binding-mutational-database-for-compu",
          "source_type": "work",
          "supports": [
            "/task_counts/total",
            "/task_counts/basis",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis"
          ]
        },
        {
          "accessed_date": "2026-08-14",
          "id": "ab-bind-automated-resource-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "737d5906df968e61eff951e1a2d47ef374b030e79cc0f3989af2332019b41d90",
            "type": "repository-path",
            "value": "official-artifact-context.json > sarahsirin/AB-Bind-Database@f5af13df80000ad9e438a6ccf1147bd5f9d55dba"
          },
          "source_id": "ab-bind-antibody-binding-mutational-database-for-compu",
          "source_type": "work",
          "supports": [
            "/resources",
            "/implementations",
            "/access/license"
          ]
        },
        {
          "accessed_date": "2026-08-14",
          "id": "ab-bind-automated-creator-evidence",
          "locator": {
            "document_page": 1,
            "note": null,
            "printed_page": "393",
            "source_fragment_sha256": "4763dbad357fb621e4a95517085090016cf28b4ece9fd9b7af581d55e15bb9c5",
            "type": "page",
            "value": "DOI"
          },
          "source_id": "ab-bind-antibody-binding-mutational-database-for-compu",
          "source_type": "work",
          "supports": [
            "/resources/0"
          ]
        },
        {
          "accessed_date": "2026-08-14",
          "id": "ab-bind-automated-version-evidence",
          "locator": {
            "document_page": 1,
            "note": null,
            "printed_page": "393",
            "source_fragment_sha256": "4763dbad357fb621e4a95517085090016cf28b4ece9fd9b7af581d55e15bb9c5",
            "type": "page",
            "value": "DOI"
          },
          "source_id": "ab-bind-antibody-binding-mutational-database-for-compu",
          "source_type": "work",
          "supports": [
            "/latest_version",
            "/versions/0"
          ]
        },
        {
          "accessed_date": "2026-08-14",
          "id": "ab-bind-automated-count-conflict-evidence",
          "locator": {
            "document_page": 5,
            "note": null,
            "printed_page": "397",
            "source_fragment_sha256": "c58a23144c26546e29f2c21e58940c1d3b67da492178a6a11b3cd2160f5e1ff4",
            "type": "section",
            "value": "Results > AB-Bind database"
          },
          "source_id": "ab-bind-antibody-binding-mutational-database-for-compu",
          "source_type": "work",
          "supports": [
            "/task_counts/subsets"
          ]
        }
      ],
      "field_status": [
        {
          "confidence": "high",
          "evidence_ids": [
            "ab-bind-automated-resource-evidence"
          ],
          "path": "/access/license",
          "reason": "The double-pass review verified the official resource identity but did not establish a redistributable benchmark license; the value remains null pending source-level license verification.",
          "status": "provisional"
        },
        {
          "confidence": "high",
          "evidence_ids": [
            "ab-bind-automated-count-conflict-evidence"
          ],
          "path": "/task_counts/subsets",
          "reason": "The owner approved the independently supported root total while all conflicted inventory subcounts were excluded from publication.",
          "status": "conflicted"
        }
      ],
      "id": "ab-bind",
      "implementations": [
        {
          "commit": "f5af13df80000ad9e438a6ccf1147bd5f9d55dba",
          "framework": "official repository",
          "notes": "Commit resolved deterministically during paper intake.",
          "status": "official",
          "url": "https://github.com/sarahsirin/AB-Bind-Database"
        }
      ],
      "kind": "dataset",
      "latest_version": "initial-release",
      "modalities": [
        "table",
        "structure-3d",
        "wet-lab-output"
      ],
      "name": "AB-Bind",
      "organizations": [
        "Massachusetts Institute of Technology",
        "Pfizer Inc."
      ],
      "parent_id": null,
      "release_date": "2015-11-06",
      "resources": [
        {
          "access_notes": "Versioned creator source; full text is not mirrored.",
          "id": "ab-bind-creator-paper-resource",
          "last_checked": "2026-08-14",
          "license": null,
          "pin": null,
          "type": "paper",
          "url": "https://doi.org/10.1002/pro.2829"
        },
        {
          "access_notes": "Official resource pinned immutably during intake.",
          "id": "ab-bind-official-repository-resource",
          "last_checked": "2026-08-14",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://github.com/sarahsirin/AB-Bind-Database/commit/f5af13df80000ad9e438a6ccf1147bd5f9d55dba",
            "value": "f5af13df80000ad9e438a6ccf1147bd5f9d55dba"
          },
          "type": "repository",
          "url": "https://github.com/sarahsirin/AB-Bind-Database"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "initial-release",
        "entries": [],
        "notes": "No Scientific Task claim passed independent high-confidence verification; task mapping remains pending a targeted official-source audit.",
        "status": "partial"
      },
      "summary": "Antibody-focused mutational binding data with accompanying structures for computational affinity prediction.",
      "task_counts": {
        "basis": "experimentally determined binding free-energy changes across the complete database",
        "reporting_status": "reported",
        "subsets": [],
        "total": 1101
      },
      "task_formats": [
        "Predict mutation-induced binding free-energy changes",
        "Classify mutants as improved or weakened binders",
        "Rank candidate mutations for improved affinity"
      ],
      "verification": {
        "last_verified": "2026-08-14",
        "notes": "New family admitted with an explicit count-inventory caveat after creator source, official resource, double-pass verification, and owner conflict resolution.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": null,
          "evidence_ids": [
            "ab-bind-automated-count-evidence",
            "ab-bind-automated-version-evidence"
          ],
          "formal_tracks": [],
          "id": "ab-bind-initial-release-version",
          "label": "initial-release",
          "notes": "Root total retained after owner review; conflicted appendix inventory subcounts are intentionally omitted.",
          "release_date": "2015-11-06",
          "status": "current",
          "task_counts": {
            "basis": "experimentally determined binding free-energy changes across the complete database",
            "reporting_status": "reported",
            "subsets": [],
            "total": 1101
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "benchmarking dataset; code repository; leaderboard",
        "biosafety_notes": null,
        "grader": "Not established by the independently verified metadata claims.",
        "level": "fully-open",
        "license": "cc-by-4.0",
        "tasks": "zero-shot affinity prediction; antibody generation"
      },
      "aliases": [
        "Antibody Binding Benchmarking"
      ],
      "audit": {
        "audited_date": "2026-08-20",
        "notes": "Automated double-pass extraction plus deterministic official-resource pin. Owner review preserved the corroborated root total and excluded conflicted appendix inventory subcounts.",
        "status": "audited-with-caveats",
        "unresolved_fields": 2
      },
      "benchmark_use_ids": [
        "abbibench-a-benchmark-for-antibody-binding-affinity-ma-abbibench-1-use",
        "abbibench-a-benchmark-for-antibody-binding-affinity-ma-abbibench-2-use",
        "explainable-protein-protein-binding-affinity-predictio-abbibench-10-use",
        "explainable-protein-protein-binding-affinity-predictio-abbibench-11-use",
        "explainable-protein-protein-binding-affinity-predictio-abbibench-12-use",
        "explainable-protein-protein-binding-affinity-predictio-abbibench-9-use"
      ],
      "capabilities": [
        "prediction",
        "design",
        "generation",
        "optimization"
      ],
      "coverage_notes": [],
      "domains": [
        "life-science",
        "protein-science",
        "protein-design",
        "protein-protein-binding",
        "antibody-antigen"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
        "explainable-protein-protein-binding-affinity-predictio"
      ],
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-automated-metadata-1-evidence",
          "locator": {
            "document_page": 16,
            "note": null,
            "printed_page": "16",
            "source_fragment_sha256": "30ca84a92b5c172c90ffbbe390d968c77bda7a900606d96eacbecbb9447ce770",
            "type": "section",
            "value": "Supplement 1"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/access/artifacts",
            "/access/level"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-automated-metadata-2-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "b529210a78e6f88303da50d14be400a61d6c31f38f6251b9f000805b28600ba2",
            "type": "repository-path",
            "value": "official-artifact-context.json: AbBibench/Antibody_Binding_Benchmark_Dataset"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/access/level"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-automated-metadata-3-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "b2d5a89a3fc1b48f18dcfe388bc5416d9add392fa8152ea3af8e3ab3c23e07e5",
            "type": "repository-path",
            "value": "official-artifact-context.json: AbBibench/Antibody_Binding_Benchmark_Dataset"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/access/license"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-automated-metadata-4-evidence",
          "locator": {
            "document_page": 6,
            "note": null,
            "printed_page": "6",
            "source_fragment_sha256": "138697f40c2416076e667d36f76b2738f59caf3691c2dce686abfe1e5556802c",
            "type": "section",
            "value": "Section 3.3"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/access/tasks",
            "/access/level"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-automated-metadata-5-evidence",
          "locator": {
            "document_page": 2,
            "note": null,
            "printed_page": "2",
            "source_fragment_sha256": "36a3d1f338dd0e75fec2389f7758e1a91cf11c7b47b019d87ddbf9e46286bfd7",
            "type": "section",
            "value": "Introduction"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/aliases"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-automated-metadata-6-evidence",
          "locator": {
            "document_page": 4,
            "note": null,
            "printed_page": "4",
            "source_fragment_sha256": "d028a1c48d1133fa841562a8cd5a757639aa28e1f7792719ffee67c58f0af9e1",
            "type": "section",
            "value": "Section 3"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/capabilities"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-automated-metadata-7-evidence",
          "locator": {
            "document_page": 2,
            "note": null,
            "printed_page": "2",
            "source_fragment_sha256": "16bc2f3a41c982f7173878c4c2d1d444e4ff2202fbacedb32ce760ced6474525",
            "type": "section",
            "value": "Introduction"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/domains"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-automated-metadata-8-evidence",
          "locator": {
            "document_page": 6,
            "note": null,
            "printed_page": "6",
            "source_fragment_sha256": "22ca3e60dad14a808aa87ace6344293593d3a62904269be708d101a8b8205018",
            "type": "section",
            "value": "Section 3.3"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/kind"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-automated-metadata-9-evidence",
          "locator": {
            "document_page": 3,
            "note": null,
            "printed_page": "3",
            "source_fragment_sha256": "4226f3a447c642cd3d60919746dd6279010fa0ed915375787bc8be84142270ad",
            "type": "figure",
            "value": "Figure 2"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/modalities"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-automated-metadata-10-evidence",
          "locator": {
            "document_page": 4,
            "note": null,
            "printed_page": "4",
            "source_fragment_sha256": "d1573386c2a0faf47499547abd1f1cdc536bdf4e1def5ee290be9efef753057e",
            "type": "section",
            "value": "Section 3"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/name"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-automated-metadata-11-evidence",
          "locator": {
            "document_page": 1,
            "note": null,
            "printed_page": "1",
            "source_fragment_sha256": "fdd02f8929f9228cf62367f6651e12a7dd42b53111a90d6319c5dee9157a08c0",
            "type": "page",
            "value": "Author affiliations"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/organizations"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-automated-metadata-12-evidence",
          "locator": {
            "document_page": 3,
            "note": null,
            "printed_page": "3",
            "source_fragment_sha256": "877577a60dd92924b8ec2368bd97107230076f2a53b46ff7c50c68c335a003b4",
            "type": "figure",
            "value": "Figure 2"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/summary"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-automated-bibliographic-evidence",
          "locator": {
            "document_page": null,
            "note": "Resolved from the canonical paper identifier during intake.",
            "printed_page": null,
            "source_fragment_sha256": "64c951cfd89a613e66a17f725b5a05fa461f67e238fec61c584a9448e08e523c",
            "type": "other",
            "value": "arXiv API bibliographic metadata"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/release_date"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-automated-count-evidence",
          "locator": {
            "document_page": 2,
            "note": null,
            "printed_page": "2",
            "source_fragment_sha256": "9715e3201460351edf1d4a4787ef97179c042b178fcf1e9a7e6ef52f538d9936",
            "type": "section",
            "value": "Introduction"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/task_counts/total",
            "/task_counts/basis",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-automated-resource-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "4120b08c51fd49fcb99a8e68faf07eae24c7bbd5bd085d0e19c38a721cdf490b",
            "type": "repository-path",
            "value": "official-artifact-context.json: AbBibench/Antibody_Binding_Benchmark_Dataset"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/resources",
            "/access/level"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-automated-creator-evidence",
          "locator": {
            "document_page": 1,
            "note": null,
            "printed_page": "1",
            "source_fragment_sha256": "86eda8ad8e4a1e6bef9c9c2ff5be1fe89b63b493f05e27e3102d5502addcba44",
            "type": "page",
            "value": "arXiv footer"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/resources/0"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-automated-version-evidence",
          "locator": {
            "document_page": 1,
            "note": null,
            "printed_page": "1",
            "source_fragment_sha256": "86eda8ad8e4a1e6bef9c9c2ff5be1fe89b63b493f05e27e3102d5502addcba44",
            "type": "page",
            "value": "arXiv footer"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/latest_version",
            "/versions/0"
          ]
        },
        {
          "accessed_date": "2026-08-20",
          "id": "abbibench-automated-count-conflict-evidence",
          "locator": {
            "document_page": 5,
            "note": null,
            "printed_page": "5",
            "source_fragment_sha256": "c13059cacfdc15232f358ec2813f65ffc2c449cf566c42198e5d67ca3307db68",
            "type": "table",
            "value": "Table 1"
          },
          "source_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
          "source_type": "work",
          "supports": [
            "/task_counts/subsets"
          ]
        }
      ],
      "field_status": [
        {
          "confidence": "medium",
          "evidence_ids": [
            "abbibench-automated-metadata-2-evidence",
            "abbibench-automated-metadata-4-evidence",
            "abbibench-automated-metadata-1-evidence",
            "abbibench-automated-resource-evidence"
          ],
          "path": "/access/level",
          "reason": "The official resource and source-located access descriptions are public, but fully-open is an Atlas-controlled classification rather than a label stated verbatim by the creator source.",
          "status": "provisional"
        },
        {
          "confidence": "high",
          "evidence_ids": [
            "abbibench-automated-count-conflict-evidence"
          ],
          "path": "/task_counts/subsets",
          "reason": "The owner approved the independently supported root total while all conflicted inventory subcounts were excluded from publication.",
          "status": "conflicted"
        }
      ],
      "id": "abbibench",
      "implementations": [],
      "kind": "suite",
      "latest_version": "initial-release",
      "modalities": [
        "protein-sequence",
        "structure-3d",
        "wet-lab-output"
      ],
      "name": "AbBiBench",
      "organizations": [
        "McWilliams School of Biomedical Informatics, UTHealth Houston",
        "Department of Industrial and Systems Engineering, Korea Advanced Institute of Science and Technology",
        "Texas Therapeutics Institute, Brown Foundation Institute of Molecular Medicine, UTHealth Houston"
      ],
      "parent_id": null,
      "release_date": "2025-05-23",
      "resources": [
        {
          "access_notes": "Versioned creator source; full text is not mirrored.",
          "id": "abbibench-creator-paper-resource",
          "last_checked": "2026-08-20",
          "license": null,
          "pin": null,
          "type": "paper",
          "url": "https://arxiv.org/abs/2506.04235"
        },
        {
          "access_notes": "Official resource pinned immutably during intake.",
          "id": "abbibench-official-dataset-resource",
          "last_checked": "2026-08-20",
          "license": "cc-by-4.0",
          "pin": {
            "kind": "commit",
            "url": "https://huggingface.co/datasets/AbBibench/Antibody_Binding_Benchmark_Dataset/tree/556fd6913aa231c0d342a8658818be8f963fd582",
            "value": "556fd6913aa231c0d342a8658818be8f963fd582"
          },
          "type": "dataset",
          "url": "https://huggingface.co/datasets/AbBibench/Antibody_Binding_Benchmark_Dataset"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "initial-release",
        "entries": [],
        "notes": "No Scientific Task claim passed independent high-confidence verification; task mapping remains pending a targeted official-source audit.",
        "status": "partial"
      },
      "summary": "A framework using antibody–antigen complexes to evaluate affinity prediction and antibody redesign.",
      "task_counts": {
        "basis": "Standardized benchmark data compiling 184,500 mutated antibodies.",
        "reporting_status": "reported",
        "subsets": [],
        "total": 184500
      },
      "task_formats": [
        "unclassified"
      ],
      "verification": {
        "last_verified": "2026-08-20",
        "notes": "New family admitted with an explicit count-inventory caveat after creator source, official resource, double-pass verification, and owner conflict resolution.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": null,
          "evidence_ids": [
            "abbibench-automated-count-evidence",
            "abbibench-automated-version-evidence"
          ],
          "formal_tracks": [],
          "id": "abbibench-initial-release-version",
          "label": "initial-release",
          "notes": "Root total retained after owner review; conflicted appendix inventory subcounts are intentionally omitted.",
          "release_date": "2025-05-23",
          "status": "current",
          "task_counts": {
            "basis": "Standardized benchmark data compiling 184,500 mutated antibodies.",
            "reporting_status": "reported",
            "subsets": [],
            "total": 184500
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "Only an aggregate model-trend chart is public.",
        "biosafety_notes": "No examples or task content are public.",
        "grader": "Not reported",
        "level": "private-internal",
        "license": null,
        "tasks": "No task content is released."
      },
      "aliases": [],
      "audit": {
        "audited_date": "2026-07-22",
        "notes": "Missing task, prompt, grader, and count fields are explicitly Not reported.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [
        "anthropic-computational-biology-evaluation"
      ],
      "capabilities": [
        "data-analysis",
        "scientific-reasoning"
      ],
      "coverage_notes": [
        {
          "count": null,
          "coverage": "explicitly-in-scope",
          "notes": "The direction is explicit; task count is not reported.",
          "reporting_status": "not_reported",
          "tag": "bioinformatics"
        }
      ],
      "domains": [
        "life-science",
        "bioinformatics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "anthropic-healthcare-life-sciences"
      ],
      "evaluation_run_ids": [
        "anthropic-computational-biology-delta"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "anthropic-computational-biology-evidence",
          "locator": {
            "note": "Exact direction label and +10.5% Opus 4.5 annotation.",
            "type": "figure",
            "value": "Evals for key life sciences tasks — Computational biology"
          },
          "source_id": "anthropic-healthcare-life-sciences",
          "source_type": "work",
          "supports": [
            "/name",
            "/summary",
            "/kind",
            "/parent_id",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/resources",
            "/implementations",
            "/versions/0/release_date",
            "/versions/0/as_of",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets"
          ]
        }
      ],
      "field_status": [],
      "id": "anthropic-computational-biology",
      "implementations": [
        {
          "commit": null,
          "framework": "Anthropic internal evaluation harness",
          "notes": "Not released.",
          "status": "not-available",
          "url": null
        }
      ],
      "kind": "track",
      "latest_version": "reported-2026-01-11",
      "modalities": [
        "text"
      ],
      "name": "Anthropic Computational Biology Eval",
      "organizations": [
        "Anthropic"
      ],
      "parent_id": "anthropic-key-life-sciences-evals",
      "release_date": "2026-01-11",
      "resources": [
        {
          "access_notes": "Official source naming this private direction and its annotated model delta.",
          "id": "anthropic-computational-biology-page-resource",
          "last_checked": "2026-07-22",
          "license": null,
          "pin": {
            "kind": "snapshot",
            "url": "https://www.anthropic.com/news/healthcare-life-sciences",
            "value": "2026-01-11-page; chart-sha256:707bbc2884cfe1ca49a419ceb053d2382d6af1819b8bb7b985423640d84fca2d"
          },
          "type": "website",
          "url": "https://www.anthropic.com/news/healthcare-life-sciences"
        }
      ],
      "scientific_task_classification": {
        "as_of": "2026-01-11",
        "benchmark_version": "reported-2026-01-11",
        "entries": [],
        "notes": "Computational biology is an explicit private direction, but the source does not identify a sufficiently specific leaf task.",
        "status": "partial"
      },
      "summary": "Private Anthropic computational-biology evaluation direction reported only through a model-trend chart, without public tasks, counts, or protocol details.",
      "task_counts": {
        "basis": "private computational biology tasks",
        "reporting_status": "not_reported",
        "subsets": [],
        "total": null
      },
      "task_formats": [
        "private computational biology evaluation; exact format not reported"
      ],
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Official direction label and exact delta only.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-01-11",
          "evidence_ids": [
            "anthropic-computational-biology-evidence"
          ],
          "formal_tracks": [],
          "id": "anthropic-computational-biology-reported-2026-01-11",
          "label": "reported-2026-01-11",
          "notes": "Metadata-only snapshot.",
          "release_date": "2026-01-11",
          "status": "current",
          "task_counts": {
            "basis": "private computational biology tasks",
            "reporting_status": "not_reported",
            "subsets": [],
            "total": null
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "Only an aggregate official chart and three direction labels are public.",
        "biosafety_notes": "No task content is public; this registry records only labels and annotated deltas and does not infer sensitive capabilities.",
        "grader": "Not reported",
        "level": "private-internal",
        "license": null,
        "tasks": "No tasks, prompts, examples, or per-item metadata are released."
      },
      "aliases": [
        "Evals for key life sciences tasks"
      ],
      "audit": {
        "audited_date": "2026-07-22",
        "notes": "All unreported settings remain null; only exact text and numeric chart annotations are normalized.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [
        "anthropic-key-life-sciences-creation"
      ],
      "capabilities": [
        "knowledge",
        "evidence-synthesis",
        "data-analysis",
        "scientific-reasoning"
      ],
      "coverage_notes": [
        {
          "count": null,
          "coverage": "explicitly-in-scope",
          "notes": "Three named internal directions are public, but tasks and counts are not.",
          "reporting_status": "not_reported",
          "tag": "life-science"
        }
      ],
      "domains": [
        "life-science",
        "bioinformatics",
        "protein-science"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "anthropic-key-life-sciences-evidence",
          "locator": {
            "note": "Names the three internal evaluation directions, labels Accuracy (%), and annotates Opus 4.5 changes from Opus 4.1.",
            "type": "figure",
            "value": "Evals for key life sciences tasks"
          },
          "source_id": "anthropic-healthcare-life-sciences",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/parent_id",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/resources",
            "/implementations",
            "/versions/0/release_date",
            "/versions/0/as_of",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/versions/0/formal_tracks"
          ]
        }
      ],
      "field_status": [],
      "id": "anthropic-key-life-sciences-evals",
      "implementations": [
        {
          "commit": null,
          "framework": "Anthropic internal evaluation harness",
          "notes": "Tasks, prompts, graders, and harness code are not released.",
          "status": "not-available",
          "url": null
        }
      ],
      "kind": "suite",
      "latest_version": "reported-2026-01-11",
      "modalities": [
        "text",
        "figure"
      ],
      "name": "Anthropic Key Life Sciences Evals",
      "organizations": [
        "Anthropic"
      ],
      "parent_id": null,
      "release_date": "2026-01-11",
      "resources": [
        {
          "access_notes": "Official page and chart reporting the private internal suite.",
          "id": "anthropic-key-life-sciences-page-resource",
          "last_checked": "2026-07-22",
          "license": null,
          "pin": {
            "kind": "snapshot",
            "url": "https://www.anthropic.com/news/healthcare-life-sciences",
            "value": "2026-01-11-page; chart-sha256:707bbc2884cfe1ca49a419ceb053d2382d6af1819b8bb7b985423640d84fca2d"
          },
          "type": "website",
          "url": "https://www.anthropic.com/news/healthcare-life-sciences"
        }
      ],
      "scientific_task_classification": {
        "as_of": "2026-01-11",
        "benchmark_version": "reported-2026-01-11",
        "entries": [],
        "notes": "The official chart names three private directions but does not disclose tasks or an exhaustive scientific taxonomy.",
        "status": "partial"
      },
      "summary": "An Anthropic private internal suite reported only through an official accuracy chart covering scientific figure interpretation, computational biology, and protein understanding.",
      "task_counts": {
        "basis": "internal evaluation tasks shown only as three directions in an official chart",
        "reporting_status": "not_reported",
        "subsets": [],
        "total": null
      },
      "task_formats": [
        "private internal accuracy evaluation; exact format not reported"
      ],
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Private/internal suite; excluded from public runnable coverage.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-01-11",
          "evidence_ids": [
            "anthropic-key-life-sciences-evidence"
          ],
          "formal_tracks": [
            "anthropic-scientific-figure-interpretation",
            "anthropic-computational-biology",
            "anthropic-protein-understanding"
          ],
          "id": "anthropic-key-life-sciences-reported-2026-01-11",
          "label": "reported-2026-01-11",
          "notes": "Metadata-only snapshot of the three directions named in the official chart.",
          "release_date": "2026-01-11",
          "status": "current",
          "task_counts": {
            "basis": "internal evaluation tasks shown only as three directions in an official chart",
            "reporting_status": "not_reported",
            "subsets": [],
            "total": null
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "Only an aggregate model-trend chart is public.",
        "biosafety_notes": "No examples or task content are public.",
        "grader": "Not reported",
        "level": "private-internal",
        "license": null,
        "tasks": "No task content is released."
      },
      "aliases": [],
      "audit": {
        "audited_date": "2026-07-22",
        "notes": "Missing task, prompt, grader, and count fields are explicitly Not reported.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [
        "anthropic-protein-understanding-evaluation"
      ],
      "capabilities": [
        "knowledge",
        "scientific-reasoning"
      ],
      "coverage_notes": [
        {
          "count": null,
          "coverage": "explicitly-in-scope",
          "notes": "The direction is explicit; task count and more specific protein task are not reported.",
          "reporting_status": "not_reported",
          "tag": "protein-science"
        }
      ],
      "domains": [
        "life-science",
        "protein-science"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "anthropic-healthcare-life-sciences"
      ],
      "evaluation_run_ids": [
        "anthropic-protein-understanding-delta"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "anthropic-protein-understanding-evidence",
          "locator": {
            "note": "Exact direction label and +10.3% Opus 4.5 annotation.",
            "type": "figure",
            "value": "Evals for key life sciences tasks — Protein understanding"
          },
          "source_id": "anthropic-healthcare-life-sciences",
          "source_type": "work",
          "supports": [
            "/name",
            "/summary",
            "/kind",
            "/parent_id",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/resources",
            "/implementations",
            "/versions/0/release_date",
            "/versions/0/as_of",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets"
          ]
        }
      ],
      "field_status": [],
      "id": "anthropic-protein-understanding",
      "implementations": [
        {
          "commit": null,
          "framework": "Anthropic internal evaluation harness",
          "notes": "Not released.",
          "status": "not-available",
          "url": null
        }
      ],
      "kind": "track",
      "latest_version": "reported-2026-01-11",
      "modalities": [
        "text"
      ],
      "name": "Anthropic Protein Understanding Eval",
      "organizations": [
        "Anthropic"
      ],
      "parent_id": "anthropic-key-life-sciences-evals",
      "release_date": "2026-01-11",
      "resources": [
        {
          "access_notes": "Official source naming this private direction and its annotated model delta.",
          "id": "anthropic-protein-understanding-page-resource",
          "last_checked": "2026-07-22",
          "license": null,
          "pin": {
            "kind": "snapshot",
            "url": "https://www.anthropic.com/news/healthcare-life-sciences",
            "value": "2026-01-11-page; chart-sha256:707bbc2884cfe1ca49a419ceb053d2382d6af1819b8bb7b985423640d84fca2d"
          },
          "type": "website",
          "url": "https://www.anthropic.com/news/healthcare-life-sciences"
        }
      ],
      "scientific_task_classification": {
        "as_of": "2026-01-11",
        "benchmark_version": "reported-2026-01-11",
        "entries": [],
        "notes": "Protein understanding is an explicit private direction, but the source does not identify a sufficiently specific leaf task.",
        "status": "partial"
      },
      "summary": "Private Anthropic protein-understanding evaluation direction reported only through a model-trend chart, without public tasks, counts, or protocol details.",
      "task_counts": {
        "basis": "private protein understanding tasks",
        "reporting_status": "not_reported",
        "subsets": [],
        "total": null
      },
      "task_formats": [
        "private protein understanding evaluation; exact format not reported"
      ],
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Official direction label and exact delta only; no leaf protein task is inferred.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-01-11",
          "evidence_ids": [
            "anthropic-protein-understanding-evidence"
          ],
          "formal_tracks": [],
          "id": "anthropic-protein-understanding-reported-2026-01-11",
          "label": "reported-2026-01-11",
          "notes": "Metadata-only snapshot.",
          "release_date": "2026-01-11",
          "status": "current",
          "task_counts": {
            "basis": "private protein understanding tasks",
            "reporting_status": "not_reported",
            "subsets": [],
            "total": null
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "Only an aggregate model-trend chart is public.",
        "biosafety_notes": "No examples or task content are public.",
        "grader": "Not reported",
        "level": "private-internal",
        "license": null,
        "tasks": "No task content is released."
      },
      "aliases": [],
      "audit": {
        "audited_date": "2026-07-22",
        "notes": "Missing task, prompt, grader, and count fields are explicitly Not reported.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [
        "anthropic-scientific-figure-evaluation"
      ],
      "capabilities": [
        "evidence-synthesis",
        "scientific-reasoning"
      ],
      "coverage_notes": [
        {
          "count": null,
          "coverage": "explicitly-in-scope",
          "notes": "The direction is explicit; task count is not reported.",
          "reporting_status": "not_reported",
          "tag": "life-science"
        }
      ],
      "domains": [
        "life-science",
        "molecular-cell-biology"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "anthropic-healthcare-life-sciences"
      ],
      "evaluation_run_ids": [
        "anthropic-scientific-figure-delta"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "anthropic-scientific-figure-evidence",
          "locator": {
            "note": "Exact direction label and +13.2% Opus 4.5 annotation.",
            "type": "figure",
            "value": "Evals for key life sciences tasks — Scientific figure interpretation"
          },
          "source_id": "anthropic-healthcare-life-sciences",
          "source_type": "work",
          "supports": [
            "/name",
            "/summary",
            "/kind",
            "/parent_id",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/resources",
            "/implementations",
            "/versions/0/release_date",
            "/versions/0/as_of",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        }
      ],
      "field_status": [],
      "id": "anthropic-scientific-figure-interpretation",
      "implementations": [
        {
          "commit": null,
          "framework": "Anthropic internal evaluation harness",
          "notes": "Not released.",
          "status": "not-available",
          "url": null
        }
      ],
      "kind": "track",
      "latest_version": "reported-2026-01-11",
      "modalities": [
        "text",
        "figure"
      ],
      "name": "Anthropic Scientific Figure Interpretation Eval",
      "organizations": [
        "Anthropic"
      ],
      "parent_id": "anthropic-key-life-sciences-evals",
      "release_date": "2026-01-11",
      "resources": [
        {
          "access_notes": "Official source naming this private direction and its annotated model delta.",
          "id": "anthropic-scientific-figure-page-resource",
          "last_checked": "2026-07-22",
          "license": null,
          "pin": {
            "kind": "snapshot",
            "url": "https://www.anthropic.com/news/healthcare-life-sciences",
            "value": "2026-01-11-page; chart-sha256:707bbc2884cfe1ca49a419ceb053d2382d6af1819b8bb7b985423640d84fca2d"
          },
          "type": "website",
          "url": "https://www.anthropic.com/news/healthcare-life-sciences"
        }
      ],
      "scientific_task_classification": {
        "as_of": "2026-01-11",
        "benchmark_version": "reported-2026-01-11",
        "entries": [
          {
            "confidence": "high",
            "count": null,
            "count_basis": "private scientific figure interpretation tasks",
            "count_ref": null,
            "count_unit": "tasks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "anthropic-scientific-figure-evidence"
            ],
            "mapping_method": "official-track",
            "notes": "No task-level examples or count are public.",
            "reporting_status": "not_reported",
            "task_type_id": "scientific-evidence-interpretation"
          }
        ],
        "notes": "The public direction label maps directly to scientific evidence interpretation; its private task count is not reported.",
        "status": "complete"
      },
      "summary": "Private Anthropic evaluation direction for scientific figure interpretation, reported only through a model-trend chart with no task count or released examples.",
      "task_counts": {
        "basis": "private scientific figure interpretation tasks",
        "reporting_status": "not_reported",
        "subsets": [],
        "total": null
      },
      "task_formats": [
        "private scientific figure interpretation; exact format not reported"
      ],
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Official direction label and exact delta only.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-01-11",
          "evidence_ids": [
            "anthropic-scientific-figure-evidence"
          ],
          "formal_tracks": [],
          "id": "anthropic-scientific-figure-reported-2026-01-11",
          "label": "reported-2026-01-11",
          "notes": "Metadata-only snapshot.",
          "release_date": "2026-01-11",
          "status": "current",
          "task_counts": {
            "basis": "private scientific figure interpretation tasks",
            "reporting_status": "not_reported",
            "subsets": [],
            "total": null
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official Python package publishes standardized LMDB loaders, split utilities, and task-specific model examples.",
        "biosafety_notes": "The suite contains public biomolecular structures and activity labels; this registry mirrors none of them.",
        "grader": "Deterministic MAE/RMSE, AUROC, accuracy, and rank-correlation evaluation depending on task.",
        "level": "fully-open",
        "license": "MIT for code; datasets retain their original source licenses and citation requirements.",
        "tasks": "All eight raw and prescribed-split datasets are publicly downloadable from the project site and Zenodo."
      },
      "aliases": [
        "Tasks On Molecules in Three Dimensions"
      ],
      "audit": {
        "audited_date": "2026-07-22",
        "notes": "Eight task identities, task semantics, public access, and package version are source-verified; dataset item counts are intentionally not summed.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction",
        "classification",
        "regression"
      ],
      "coverage_notes": [
        {
          "count": 1,
          "coverage": "explicitly-in-scope",
          "notes": "Protein Interface Prediction is one of the eight official datasets.",
          "reporting_status": "reported",
          "tag": "protein-protein-binding"
        },
        {
          "count": 2,
          "coverage": "explicitly-in-scope",
          "notes": "Ligand Binding Affinity and Ligand Efficacy Prediction are distinct official datasets.",
          "reporting_status": "reported",
          "tag": "protein-ligand-binding"
        },
        {
          "count": 3,
          "coverage": "explicitly-in-scope",
          "notes": "Residue Identity, Mutation Stability Prediction, and Protein Structure Ranking operate on protein structures; their task semantics remain distinct.",
          "reporting_status": "reported",
          "tag": "protein-structure"
        }
      ],
      "domains": [
        "protein-structure",
        "protein-protein-binding",
        "protein-ligand-binding",
        "medchem"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "atom3d-paper"
      ],
      "evaluation_run_ids": [
        "atom3d-creator-full"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "atom3d-paper-definition-evidence",
          "locator": {
            "note": "Defines SMP, PIP, RES, MSP, LBA, LEP, PSR, and RSR and their task-specific evaluation.",
            "type": "table",
            "value": "Sections 2-4 and Tables 1-2"
          },
          "source_id": "atom3d-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/resources",
            "/versions/0/release_date",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0",
            "/scientific_task_classification/entries/1",
            "/scientific_task_classification/entries/2",
            "/scientific_task_classification/entries/3",
            "/scientific_task_classification/entries/4",
            "/scientific_task_classification/entries/5",
            "/scientific_task_classification/entries/6",
            "/scientific_task_classification/entries/7"
          ]
        },
        {
          "accessed_date": "2026-07-22",
          "id": "atom3d-project-evidence",
          "locator": {
            "note": "Confirms the eight-task living catalog and gives exact task, split, download, and license statements for each dataset.",
            "type": "web-anchor",
            "value": "Datasets; SMP, PIP, RES, MSP, LBA, LEP, PSR, and RSR detail pages"
          },
          "source_id": "atom3d-project-resource",
          "source_type": "resource",
          "supports": [
            "/summary",
            "/domains",
            "/modalities",
            "/task_counts/total",
            "/coverage_notes",
            "/access/level",
            "/access/tasks",
            "/access/grader",
            "/access/license",
            "/resources",
            "/versions/0/as_of"
          ]
        },
        {
          "accessed_date": "2026-07-22",
          "id": "atom3d-repository-evidence",
          "locator": {
            "note": "Confirms package version, public loaders, eight task implementations, and MIT license.",
            "type": "repository-path",
            "value": "README.md; atom3d/__init__.py; examples/{smp,pip,res,msp,lba,lep,psr,rsr} at 4c2f3b7e9efe128791b83f03b2e8cae91e78b018"
          },
          "source_id": "atom3d-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/resources",
            "/implementations",
            "/versions/0/as_of",
            "/versions/0/formal_tracks"
          ]
        }
      ],
      "field_status": [],
      "id": "atom3d",
      "implementations": [
        {
          "commit": "4c2f3b7e9efe128791b83f03b2e8cae91e78b018",
          "framework": "atom3d Python package",
          "notes": "LMDB data tooling and task-specific CNN, GNN, and equivariant-network examples.",
          "status": "official",
          "url": "https://github.com/drorlab/atom3d/tree/4c2f3b7e9efe128791b83f03b2e8cae91e78b018"
        }
      ],
      "kind": "suite",
      "latest_version": "v0.2.6",
      "modalities": [
        "structure-3d"
      ],
      "name": "ATOM3D",
      "organizations": [
        "Stanford University"
      ],
      "parent_id": null,
      "release_date": "2020-12-07",
      "resources": [
        {
          "access_notes": "NeurIPS 2021 Datasets and Benchmarks creator paper.",
          "id": "atom3d-paper-resource",
          "last_checked": "2026-07-22",
          "license": null,
          "pin": null,
          "type": "paper",
          "url": "https://datasets-benchmarks-proceedings.neurips.cc/paper/2021/hash/c45147dee729311ef5b5c3003946c48f-Abstract-round1.html"
        },
        {
          "access_notes": "Official task catalog with task definitions, splits, downloads, and dataset-specific license statements.",
          "id": "atom3d-project-resource",
          "last_checked": "2026-07-22",
          "license": null,
          "pin": {
            "kind": "snapshot",
            "url": "https://www.atom3d.ai/",
            "value": "2026-07-22"
          },
          "type": "website",
          "url": "https://www.atom3d.ai/"
        },
        {
          "access_notes": "Official package and task examples; package version v0.2.6.",
          "id": "atom3d-repository-resource",
          "last_checked": "2026-07-22",
          "license": "MIT",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/drorlab/atom3d/tree/4c2f3b7e9efe128791b83f03b2e8cae91e78b018",
            "value": "4c2f3b7e9efe128791b83f03b2e8cae91e78b018"
          },
          "type": "repository",
          "url": "https://github.com/drorlab/atom3d"
        }
      ],
      "scientific_task_classification": {
        "as_of": "2026-07-22",
        "benchmark_version": "v0.2.6",
        "entries": [
          {
            "confidence": "high",
            "count": 1,
            "count_basis": "Curated 3D benchmark datasets.",
            "count_ref": null,
            "count_unit": "tasks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "atom3d-paper-definition-evidence"
            ],
            "mapping_method": "official-track",
            "notes": "SMP, small-molecule property prediction from molecular structure.",
            "reporting_status": "reported",
            "task_type_id": "small-molecule-property-prediction"
          },
          {
            "confidence": "high",
            "count": 1,
            "count_basis": "Curated 3D benchmark datasets.",
            "count_ref": null,
            "count_unit": "tasks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "atom3d-paper-definition-evidence"
            ],
            "mapping_method": "official-track",
            "notes": "PIP predicts whether residue pairs from two proteins contact when the proteins bind; it is narrower than binary interaction detection.",
            "reporting_status": "reported",
            "task_type_id": "protein-protein-interface-prediction"
          },
          {
            "confidence": "high",
            "count": 1,
            "count_basis": "Curated 3D benchmark datasets.",
            "count_ref": null,
            "count_unit": "tasks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "atom3d-paper-definition-evidence"
            ],
            "mapping_method": "official-track",
            "notes": "RES, residue identity prediction from the local structural environment.",
            "reporting_status": "reported",
            "task_type_id": "protein-residue-identity-prediction"
          },
          {
            "confidence": "high",
            "count": 1,
            "count_basis": "Curated 3D benchmark datasets.",
            "count_ref": null,
            "count_unit": "tasks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "atom3d-paper-definition-evidence"
            ],
            "mapping_method": "official-track",
            "notes": "MSP predicts whether a single mutation stabilizes a protein complex; it is not a monomer thermostability task.",
            "reporting_status": "reported",
            "task_type_id": "protein-complex-mutation-stability-prediction"
          },
          {
            "confidence": "high",
            "count": 1,
            "count_basis": "Curated 3D benchmark datasets.",
            "count_ref": null,
            "count_unit": "tasks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "atom3d-paper-definition-evidence"
            ],
            "mapping_method": "official-track",
            "notes": "LBA, ligand binding-affinity prediction.",
            "reporting_status": "reported",
            "task_type_id": "protein-ligand-binding-affinity"
          },
          {
            "confidence": "high",
            "count": 1,
            "count_basis": "Curated 3D benchmark datasets.",
            "count_ref": null,
            "count_unit": "tasks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "atom3d-paper-definition-evidence"
            ],
            "mapping_method": "official-track",
            "notes": "LEP classifies whether a bound molecule activates protein function from active and inactive target conformations; it does not predict binding presence or affinity.",
            "reporting_status": "reported",
            "task_type_id": "protein-ligand-efficacy-prediction"
          },
          {
            "confidence": "high",
            "count": 1,
            "count_basis": "Curated 3D benchmark datasets.",
            "count_ref": null,
            "count_unit": "tasks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "atom3d-paper-definition-evidence"
            ],
            "mapping_method": "official-track",
            "notes": "PSR, protein structure ranking.",
            "reporting_status": "reported",
            "task_type_id": "protein-model-quality-assessment"
          },
          {
            "confidence": "high",
            "count": 1,
            "count_basis": "Curated 3D benchmark datasets.",
            "count_ref": null,
            "count_unit": "tasks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "atom3d-paper-definition-evidence"
            ],
            "mapping_method": "official-track",
            "notes": "RSR ranks candidate RNA structures and is therefore not relabeled as de novo RNA folding.",
            "reporting_status": "reported",
            "task_type_id": "rna-structure-quality-assessment"
          }
        ],
        "notes": "The eight official task abbreviations are mapped one-to-one to their documented scientific prediction problems.",
        "status": "complete"
      },
      "summary": "A living collection of eight curated 3D molecular-learning tasks spanning small molecules, protein interactions and mutations, ligand binding, and protein/RNA structure ranking.",
      "task_counts": {
        "basis": "curated 3D benchmark datasets",
        "reporting_status": "reported",
        "subsets": [],
        "total": 8
      },
      "task_formats": [
        "3D molecular regression",
        "3D molecular classification",
        "residue classification",
        "structure ranking"
      ],
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "The paper, project site, documentation, and pinned package agree on eight tasks.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-07-22",
          "evidence_ids": [
            "atom3d-paper-definition-evidence",
            "atom3d-project-evidence",
            "atom3d-repository-evidence"
          ],
          "formal_tracks": [],
          "id": "atom3d-v0-2-6",
          "label": "v0.2.6",
          "notes": "As-of snapshot of the eight datasets listed by the living project site and package v0.2.6.",
          "release_date": "2020-12-07",
          "status": "rolling",
          "task_counts": {
            "basis": "curated 3D benchmark datasets",
            "reporting_status": "reported",
            "subsets": [],
            "total": 8
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "Training code, tokenizers, RNA language-model adapters, checkpoints, and split-specific scripts are public.",
        "biosafety_notes": "The paper notes general dual-use potential for RNA manipulation; this registry mirrors no sequences, checkpoints, or model outputs.",
        "grader": "Deterministic task-specific F1, precision, R-squared, accuracy, AUC, MCRMSE, and Spearman scorers.",
        "level": "fully-open",
        "license": "Apache-2.0 for code; underlying datasets have mixed public, research-only, and noncommercial terms documented in the paper.",
        "tasks": "The official repository links all 13 prepared datasets and publishes task scripts and model checkpoints."
      },
      "aliases": [
        "Benchmark for Comprehensive RNA Tasks and Language Models",
        "RNA BEACON"
      ],
      "audit": {
        "audited_date": "2026-07-22",
        "notes": "The 13 task rows, group counts, split sizes, metrics, repetition count, access, and dataset-license table were checked.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction",
        "classification",
        "regression"
      ],
      "coverage_notes": [
        {
          "count": 13,
          "coverage": "explicitly-in-scope",
          "notes": "All tasks consume RNA sequence and evaluate RNA structural, functional, or engineering outcomes.",
          "reporting_status": "reported",
          "tag": "transcriptomics"
        },
        {
          "count": 2,
          "coverage": "explicitly-in-scope",
          "notes": "CRISPR on-target and off-target tasks involve guide and target sequence activity.",
          "reporting_status": "reported",
          "tag": "genomics"
        }
      ],
      "domains": [
        "transcriptomics",
        "genomics",
        "molecular-cell-biology"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "beacon-paper"
      ],
      "evaluation_run_ids": [
        "beacon-creator-full"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "beacon-paper-definition-evidence",
          "locator": {
            "note": "Defines all 13 tasks, split sizes, metrics, three-seed evaluation, and dataset-specific licenses.",
            "type": "table",
            "value": "pp. 1-7, Figure 1 and Tables 1-3; Appendix A.1 and Table 22"
          },
          "source_id": "beacon-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/access/biosafety_notes",
            "/resources",
            "/versions/0/release_date",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0",
            "/scientific_task_classification/entries/1",
            "/scientific_task_classification/entries/2",
            "/scientific_task_classification/entries/3",
            "/scientific_task_classification/entries/4",
            "/scientific_task_classification/entries/5",
            "/scientific_task_classification/entries/6",
            "/scientific_task_classification/entries/7"
          ]
        },
        {
          "accessed_date": "2026-07-22",
          "id": "beacon-repository-evidence",
          "locator": {
            "note": "Confirms exact task names, public implementation, task-specific scorers, and Apache-2.0 code license.",
            "type": "repository-path",
            "value": "README.md; downstream/*; scripts/* at da7f9c7ac3f39605af27e1dfcdf879adba963d79"
          },
          "source_id": "beacon-repository-resource",
          "source_type": "resource",
          "supports": [
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/implementations",
            "/versions/0/formal_tracks"
          ]
        }
      ],
      "field_status": [],
      "id": "beacon-rna",
      "implementations": [
        {
          "commit": "da7f9c7ac3f39605af27e1dfcdf879adba963d79",
          "framework": "BEACON RNABenchmark",
          "notes": "Scripts for all 13 tasks and BEACON-B/open-source RNA model evaluation.",
          "status": "official",
          "url": "https://github.com/terry-r123/RNABenchmark/tree/da7f9c7ac3f39605af27e1dfcdf879adba963d79"
        }
      ],
      "kind": "suite",
      "latest_version": "neurips-2024",
      "modalities": [
        "dna-rna-sequence"
      ],
      "name": "BEACON",
      "organizations": [
        "Shanghai Artificial Intelligence Laboratory",
        "University of Sydney",
        "University of Hong Kong",
        "Fudan University"
      ],
      "parent_id": null,
      "release_date": "2024-06-14",
      "resources": [
        {
          "access_notes": "NeurIPS 2024 Datasets and Benchmarks creator paper.",
          "id": "beacon-paper-resource",
          "last_checked": "2026-07-22",
          "license": null,
          "pin": null,
          "type": "paper",
          "url": "https://proceedings.neurips.cc/paper_files/paper/2024/file/a8ea503d91320fcfe12cba61c8a6d285-Paper-Datasets_and_Benchmarks_Track.pdf"
        },
        {
          "access_notes": "Official implementation and task manifest.",
          "id": "beacon-repository-resource",
          "last_checked": "2026-07-22",
          "license": "Apache-2.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/terry-r123/RNABenchmark/tree/da7f9c7ac3f39605af27e1dfcdf879adba963d79",
            "value": "da7f9c7ac3f39605af27e1dfcdf879adba963d79"
          },
          "type": "repository",
          "url": "https://github.com/terry-r123/RNABenchmark"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "neurips-2024",
        "entries": [
          {
            "confidence": "high",
            "count": 4,
            "count_basis": "Formal RNA benchmark tasks.",
            "count_ref": null,
            "count_unit": "tasks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "beacon-paper-definition-evidence"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "Secondary structure, structural score, distance-map, and tertiary-structure tasks.",
            "reporting_status": "reported",
            "task_type_id": "rna-structure-prediction"
          },
          {
            "confidence": "high",
            "count": 3,
            "count_basis": "Formal RNA benchmark tasks.",
            "count_ref": null,
            "count_unit": "tasks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "beacon-paper-definition-evidence"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "Splice-site prediction, alternative polyadenylation, and mean ribosome loading.",
            "reporting_status": "reported",
            "task_type_id": "rna-processing-translation-prediction"
          },
          {
            "confidence": "high",
            "count": 1,
            "count_basis": "Formal RNA benchmark tasks.",
            "count_ref": null,
            "count_unit": "tasks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "beacon-paper-definition-evidence"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "Non-coding RNA functional classification.",
            "reporting_status": "reported",
            "task_type_id": "rna-function-classification"
          },
          {
            "confidence": "high",
            "count": 1,
            "count_basis": "Formal RNA benchmark tasks.",
            "count_ref": null,
            "count_unit": "tasks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "beacon-paper-definition-evidence"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "RNA modification-site prediction.",
            "reporting_status": "reported",
            "task_type_id": "rna-modification-prediction"
          },
          {
            "confidence": "high",
            "count": 1,
            "count_basis": "Formal RNA benchmark tasks.",
            "count_ref": null,
            "count_unit": "tasks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "beacon-paper-definition-evidence"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "Condition-specific RNA degradation prediction.",
            "reporting_status": "reported",
            "task_type_id": "rna-stability-degradation-prediction"
          },
          {
            "confidence": "high",
            "count": 1,
            "count_basis": "Formal RNA benchmark tasks.",
            "count_ref": null,
            "count_unit": "tasks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "beacon-paper-definition-evidence"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "Programmable RNA-switch activity prediction.",
            "reporting_status": "reported",
            "task_type_id": "rna-switch-activity-prediction"
          },
          {
            "confidence": "high",
            "count": 1,
            "count_basis": "Formal RNA benchmark tasks.",
            "count_ref": null,
            "count_unit": "tasks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "beacon-paper-definition-evidence"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "CRISPR on-target efficiency prediction.",
            "reporting_status": "reported",
            "task_type_id": "crispr-guide-activity-prediction"
          },
          {
            "confidence": "high",
            "count": 1,
            "count_basis": "Formal RNA benchmark tasks.",
            "count_ref": null,
            "count_unit": "tasks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "beacon-paper-definition-evidence"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "CRISPR off-target interaction prediction.",
            "reporting_status": "reported",
            "task_type_id": "crispr-off-target-prediction"
          }
        ],
        "notes": "All thirteen tasks are assigned once using the structure, function, and engineering groupings in the creator paper.",
        "status": "complete"
      },
      "summary": "A 13-task RNA representation benchmark covering structure, function, processing, modification, translation, degradation, programmable switches, and CRISPR activity.",
      "task_counts": {
        "basis": "formal RNA benchmark tasks",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "formal RNA benchmark tasks",
            "count": 4,
            "exclusive": true,
            "exhaustive": true,
            "id": "beacon-structure",
            "label": "Structure tasks",
            "notes": "SSP, CMP, DMP, and SSI.",
            "reporting_status": "reported"
          },
          {
            "basis": "formal RNA benchmark tasks",
            "count": 5,
            "exclusive": true,
            "exhaustive": true,
            "id": "beacon-function",
            "label": "Function tasks",
            "notes": "SPL, APA, ncRNA, Modif, and MRL.",
            "reporting_status": "reported"
          },
          {
            "basis": "formal RNA benchmark tasks",
            "count": 4,
            "exclusive": true,
            "exhaustive": true,
            "id": "beacon-engineering",
            "label": "Engineering tasks",
            "notes": "VDP, PRS, CRI-On, and CRI-Off.",
            "reporting_status": "reported"
          }
        ],
        "total": 13
      },
      "task_formats": [
        "sequence classification",
        "nucleotide-level classification",
        "sequence regression",
        "nucleotide-level regression"
      ],
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Paper Table 1 and the pinned repository task manifest agree.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": null,
          "evidence_ids": [
            "beacon-paper-definition-evidence",
            "beacon-repository-evidence"
          ],
          "formal_tracks": [],
          "id": "beacon-neurips-2024",
          "label": "neurips-2024",
          "notes": "Fixed 13-task snapshot from the final NeurIPS paper.",
          "release_date": "2024-06-14",
          "status": "current",
          "task_counts": {
            "basis": "formal RNA benchmark tasks",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "formal RNA benchmark tasks",
                "count": 4,
                "exclusive": true,
                "exhaustive": true,
                "id": "beacon-structure",
                "label": "Structure tasks",
                "notes": "SSP, CMP, DMP, and SSI.",
                "reporting_status": "reported"
              },
              {
                "basis": "formal RNA benchmark tasks",
                "count": 5,
                "exclusive": true,
                "exhaustive": true,
                "id": "beacon-function",
                "label": "Function tasks",
                "notes": "SPL, APA, ncRNA, Modif, and MRL.",
                "reporting_status": "reported"
              },
              {
                "basis": "formal RNA benchmark tasks",
                "count": 4,
                "exclusive": true,
                "exhaustive": true,
                "id": "beacon-engineering",
                "label": "Engineering tasks",
                "notes": "VDP, PRS, CRI-On, and CRI-Off.",
                "reporting_status": "reported"
              }
            ],
            "total": 13
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "See the linked official creator resources.",
        "biosafety_notes": null,
        "grader": "Not established by the independently verified metadata claims.",
        "level": "fully-open",
        "license": null,
        "tasks": "See the linked official creator resources."
      },
      "aliases": [
        "dual-guide benchmark library",
        "dual-guide benchmarking library"
      ],
      "audit": {
        "audited_date": "2026-08-02",
        "notes": "Automated double-pass extraction plus deterministic official-resource pin; the benchmark license remains unverified and is published as null.",
        "status": "audited-with-caveats",
        "unresolved_fields": 2
      },
      "benchmark_use_ids": [
        "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-benchmark-dual-human-crispr-cas9-library-3-use",
        "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-benchmark-dual-human-crispr-cas9-library-4-use"
      ],
      "capabilities": [
        "prediction"
      ],
      "coverage_notes": [],
      "domains": [
        "genomics",
        "molecular-cell-biology",
        "assay-screening"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "a-benchmark-comparison-of-crisprn-guide-rna-design-alg"
      ],
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-08-02",
          "id": "benchmark-dual-human-crispr-cas9-library-automated-metadata-1-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "6badb7f86f61b6567302684be94b7d02ad77de557ce9bb92d14e2ae97b073523",
            "type": "section",
            "value": "Data availability"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/access/level"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "benchmark-dual-human-crispr-cas9-library-automated-metadata-2-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "b6beacdabfa3da634d1c5ed5956190f955ca94ec97d7c20c431efc55dca0fa95",
            "type": "section",
            "value": "Figure 2 caption and Methods, Sec9, Par26"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/aliases"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "benchmark-dual-human-crispr-cas9-library-automated-metadata-3-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "823dfc9bdf7e718c2f90d05308b5720643d904f878b55ade7bbbc0c2433225de",
            "type": "section",
            "value": "Methods, Sec9, Par26; Results and discussion, Par12"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/capabilities"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "benchmark-dual-human-crispr-cas9-library-automated-metadata-4-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "62a256c216c566e7bdb30cb276eb0e471d6bf2b1eff4710200a8e0549950cbb0",
            "type": "section",
            "value": "Abstract, Par9"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/domains"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "benchmark-dual-human-crispr-cas9-library-automated-metadata-5-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "a348a05ccd30032681a84d7b2a72b38aba6ef8e77dfdc9b55049e5643b4b693a",
            "type": "section",
            "value": "Data availability"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/kind"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "benchmark-dual-human-crispr-cas9-library-automated-metadata-6-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "93de163bffb3720ce82dd25f33333bec6180cb488277c2efd3ec9548a11cff37",
            "type": "section",
            "value": "Data availability"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/modalities"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "benchmark-dual-human-crispr-cas9-library-automated-metadata-7-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "874d21c535461e8fe020d585de0c412948736c9c14c902474c8ec11f69de00c8",
            "type": "section",
            "value": "Results and discussion, Par14"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/name"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "benchmark-dual-human-crispr-cas9-library-automated-metadata-8-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "dd142f02521050383739d43e085b5bd2f3c741c734b04acced8aae31e800e9e7",
            "type": "section",
            "value": "Front matter: contrib-group and Aff1"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/organizations"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "benchmark-dual-human-crispr-cas9-library-automated-metadata-9-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "7bd819988af194ec507f63842d57e20ab9b73f76a63789e9d2609260c14a9e18",
            "type": "section",
            "value": "Results and discussion, Par14"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/summary"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "benchmark-dual-human-crispr-cas9-library-automated-bibliographic-evidence",
          "locator": {
            "document_page": null,
            "note": "Resolved from the canonical paper identifier during intake.",
            "printed_page": null,
            "source_fragment_sha256": "b5c780106941b5f895588a9efedaafe8aeea464cff6edb5121894832c0978052",
            "type": "other",
            "value": "Crossref bibliographic metadata"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/release_date"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "benchmark-dual-human-crispr-cas9-library-automated-count-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "cca1499a2c0db292e40ad81a43e9da5edd090862b9394ff1ce9c751691ac1704",
            "type": "section",
            "value": "Methods, Sec9, Par26"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/task_counts/total",
            "/task_counts/basis",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "benchmark-dual-human-crispr-cas9-library-automated-subset-coverage-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "cca1499a2c0db292e40ad81a43e9da5edd090862b9394ff1ce9c751691ac1704",
            "type": "section",
            "value": "Methods, Sec9, Par26"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/task_counts/subsets",
            "/versions/0/task_counts/subsets"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "benchmark-dual-human-crispr-cas9-library-automated-resource-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "6badb7f86f61b6567302684be94b7d02ad77de557ce9bb92d14e2ae97b073523",
            "type": "section",
            "value": "Data availability"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/resources",
            "/access/license"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "benchmark-dual-human-crispr-cas9-library-automated-creator-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "fbdde9f192cdcc4c69a769fe8e89efec9636b795fdba90d128d55722bb45aa80",
            "type": "section",
            "value": "Front matter: article DOI"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/resources/0"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "benchmark-dual-human-crispr-cas9-library-automated-version-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "fbdde9f192cdcc4c69a769fe8e89efec9636b795fdba90d128d55722bb45aa80",
            "type": "section",
            "value": "Front matter: article DOI"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/latest_version",
            "/versions/0"
          ]
        }
      ],
      "field_status": [
        {
          "confidence": "high",
          "evidence_ids": [
            "benchmark-dual-human-crispr-cas9-library-automated-subset-coverage-evidence"
          ],
          "path": "/task_counts/subsets",
          "reason": "The double-pass review established the root item total but did not establish an exhaustive formal-subset inventory; the empty list must not be interpreted as evidence that the benchmark has no subsets.",
          "status": "provisional"
        },
        {
          "confidence": "high",
          "evidence_ids": [
            "benchmark-dual-human-crispr-cas9-library-automated-resource-evidence"
          ],
          "path": "/access/license",
          "reason": "The double-pass review verified the official resource identity but did not establish a redistributable benchmark license; the value remains null pending source-level license verification.",
          "status": "provisional"
        }
      ],
      "id": "benchmark-dual-human-crispr-cas9-library",
      "implementations": [],
      "kind": "dataset",
      "latest_version": "initial-release",
      "modalities": [
        "dna-rna-sequence",
        "wet-lab-output"
      ],
      "name": "benchmark-dual human CRISPR-Cas9 library",
      "organizations": [
        "Joint Astrazeneca-Cancer Research Horizons Functional Genomics Centre"
      ],
      "parent_id": null,
      "release_date": "2025-02-26",
      "resources": [
        {
          "access_notes": "Versioned creator source; full text is not mirrored.",
          "id": "benchmark-dual-human-crispr-cas9-library-creator-paper-resource",
          "last_checked": "2026-08-02",
          "license": null,
          "pin": null,
          "type": "paper",
          "url": "https://doi.org/10.1186/s12864-025-11386-3"
        },
        {
          "access_notes": "Official resource pinned immutably during intake.",
          "id": "benchmark-dual-human-crispr-cas9-library-official-dataset-resource",
          "last_checked": "2026-08-02",
          "license": null,
          "pin": {
            "kind": "version",
            "url": "https://zenodo.org/records/11164566",
            "value": "1.0.0"
          },
          "type": "dataset",
          "url": "https://zenodo.org/records/11164566"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "initial-release",
        "entries": [],
        "notes": "Candidate artifact-derived Scientific Task mappings lacked an official artifact-level locator and were omitted; task mapping remains pending a targeted official-artifact audit.",
        "status": "partial"
      },
      "summary": "A human CRISPR-Cas9 paired-guide library created to compare dual- and single-targeting strategies in loss-of-function screens.",
      "task_counts": {
        "basis": "Source-reported final dual-guide benchmarking library inventory.",
        "reporting_status": "reported",
        "subsets": [],
        "total": 45720
      },
      "task_formats": [
        "unclassified"
      ],
      "verification": {
        "last_verified": "2026-08-02",
        "notes": "New family admitted after creator source and official resource verification; the unresolved benchmark license is visibly flagged.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": null,
          "evidence_ids": [
            "benchmark-dual-human-crispr-cas9-library-automated-count-evidence",
            "benchmark-dual-human-crispr-cas9-library-automated-version-evidence"
          ],
          "formal_tracks": [],
          "id": "benchmark-dual-human-crispr-cas9-library-initial-release-version",
          "label": "initial-release",
          "notes": "Atlas snapshot label for the creator-paper release; the source does not report a formal benchmark version.",
          "release_date": "2025-02-26",
          "status": "current",
          "task_counts": {
            "basis": "Source-reported final dual-guide benchmarking library inventory.",
            "reporting_status": "reported",
            "subsets": [],
            "total": 45720
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "See the linked official creator resources.",
        "biosafety_notes": null,
        "grader": "Not established by the independently verified metadata claims.",
        "level": "fully-open",
        "license": null,
        "tasks": "See the linked official creator resources."
      },
      "aliases": [],
      "audit": {
        "audited_date": "2026-08-02",
        "notes": "Automated double-pass extraction plus deterministic official-resource pin; the benchmark license remains unverified and is published as null.",
        "status": "audited-with-caveats",
        "unresolved_fields": 2
      },
      "benchmark_use_ids": [
        "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-benchmark-human-crispr-cas9-library-1-use",
        "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-benchmark-human-crispr-cas9-library-2-use"
      ],
      "capabilities": [
        "prediction"
      ],
      "coverage_notes": [],
      "domains": [
        "genomics",
        "molecular-cell-biology",
        "assay-screening"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "a-benchmark-comparison-of-crisprn-guide-rna-design-alg"
      ],
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-08-02",
          "id": "benchmark-human-crispr-cas9-library-automated-metadata-1-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "6badb7f86f61b6567302684be94b7d02ad77de557ce9bb92d14e2ae97b073523",
            "type": "section",
            "value": "Data availability"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/access/level"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "benchmark-human-crispr-cas9-library-automated-metadata-2-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "13167289ab06df30a0424eecfadbe79ec8adfd3b9d568f0a36aea26cd90e9f57",
            "type": "section",
            "value": "Results and discussion, Par12"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/capabilities"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "benchmark-human-crispr-cas9-library-automated-metadata-3-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "eb939fb1d005b66196245f0a75394947085d00f2090155358268e0f25e863464",
            "type": "section",
            "value": "Abstract, Par9"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/domains"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "benchmark-human-crispr-cas9-library-automated-metadata-4-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "a348a05ccd30032681a84d7b2a72b38aba6ef8e77dfdc9b55049e5643b4b693a",
            "type": "section",
            "value": "Data availability"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/kind"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "benchmark-human-crispr-cas9-library-automated-metadata-5-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "6d3a7cb815a9ba0e415b6dc6412fbd9d18c4c12151db2a29dced5b471cbff731",
            "type": "section",
            "value": "Results and discussion, Par11"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/modalities"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "benchmark-human-crispr-cas9-library-automated-metadata-6-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "24e393d6f1c90f63001895a1ef5f8c5ef57d6e391c7ca753e172580946d53eb7",
            "type": "section",
            "value": "Results and discussion, Par11"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/name"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "benchmark-human-crispr-cas9-library-automated-metadata-7-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "dd142f02521050383739d43e085b5bd2f3c741c734b04acced8aae31e800e9e7",
            "type": "section",
            "value": "Front matter: contrib-group and Aff1"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/organizations"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "benchmark-human-crispr-cas9-library-automated-metadata-8-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "652b34a951af118b76a7a8e39b41ea6acf996813a8b7bcaa20e093732222e8a8",
            "type": "section",
            "value": "Results and discussion, Par11"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/summary"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "benchmark-human-crispr-cas9-library-automated-bibliographic-evidence",
          "locator": {
            "document_page": null,
            "note": "Resolved from the canonical paper identifier during intake.",
            "printed_page": null,
            "source_fragment_sha256": "b5c780106941b5f895588a9efedaafe8aeea464cff6edb5121894832c0978052",
            "type": "other",
            "value": "Crossref bibliographic metadata"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/release_date"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "benchmark-human-crispr-cas9-library-automated-count-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "2da24a2f514c4e793058a6cce4a9b95fc66247d6c88b45347381d2a203fc4ad0",
            "type": "section",
            "value": "Results and discussion, Par11"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/task_counts/total",
            "/task_counts/basis",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "benchmark-human-crispr-cas9-library-automated-subset-coverage-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "2da24a2f514c4e793058a6cce4a9b95fc66247d6c88b45347381d2a203fc4ad0",
            "type": "section",
            "value": "Results and discussion, Par11"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/task_counts/subsets",
            "/versions/0/task_counts/subsets"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "benchmark-human-crispr-cas9-library-automated-resource-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "6badb7f86f61b6567302684be94b7d02ad77de557ce9bb92d14e2ae97b073523",
            "type": "section",
            "value": "Data availability"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/resources",
            "/access/license"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "benchmark-human-crispr-cas9-library-automated-creator-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "fbdde9f192cdcc4c69a769fe8e89efec9636b795fdba90d128d55722bb45aa80",
            "type": "section",
            "value": "Front matter: article DOI"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/resources/0"
          ]
        },
        {
          "accessed_date": "2026-08-02",
          "id": "benchmark-human-crispr-cas9-library-automated-version-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "fbdde9f192cdcc4c69a769fe8e89efec9636b795fdba90d128d55722bb45aa80",
            "type": "section",
            "value": "Front matter: article DOI"
          },
          "source_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
          "source_type": "work",
          "supports": [
            "/latest_version",
            "/versions/0"
          ]
        }
      ],
      "field_status": [
        {
          "confidence": "high",
          "evidence_ids": [
            "benchmark-human-crispr-cas9-library-automated-subset-coverage-evidence"
          ],
          "path": "/task_counts/subsets",
          "reason": "The double-pass review did not establish an exhaustive formal-subset inventory; the empty list must not be interpreted as evidence that the benchmark has no subsets.",
          "status": "provisional"
        },
        {
          "confidence": "high",
          "evidence_ids": [
            "benchmark-human-crispr-cas9-library-automated-resource-evidence"
          ],
          "path": "/access/license",
          "reason": "The double-pass review verified the official resource identity but did not establish a redistributable benchmark license; the value remains null pending source-level license verification.",
          "status": "provisional"
        }
      ],
      "id": "benchmark-human-crispr-cas9-library",
      "implementations": [],
      "kind": "dataset",
      "latest_version": "initial-release",
      "modalities": [
        "dna-rna-sequence",
        "wet-lab-output"
      ],
      "name": "benchmark human CRISPR-Cas9 library",
      "organizations": [
        "Joint Astrazeneca-Cancer Research Horizons Functional Genomics Centre"
      ],
      "parent_id": null,
      "release_date": "2025-02-26",
      "resources": [
        {
          "access_notes": "Versioned creator source; full text is not mirrored.",
          "id": "benchmark-human-crispr-cas9-library-creator-paper-resource",
          "last_checked": "2026-08-02",
          "license": null,
          "pin": null,
          "type": "paper",
          "url": "https://doi.org/10.1186/s12864-025-11386-3"
        },
        {
          "access_notes": "Official resource pinned immutably during intake.",
          "id": "benchmark-human-crispr-cas9-library-official-dataset-resource",
          "last_checked": "2026-08-02",
          "license": null,
          "pin": {
            "kind": "version",
            "url": "https://zenodo.org/records/11164566",
            "value": "1.0.0"
          },
          "type": "dataset",
          "url": "https://zenodo.org/records/11164566"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "initial-release",
        "entries": [],
        "notes": "Candidate artifact-derived Scientific Task mappings lacked an official artifact-level locator and were omitted; task mapping remains pending a targeted official-artifact audit.",
        "status": "partial"
      },
      "summary": "A human CRISPR-Cas9 guide-RNA library assembled to compare single-targeting library performance in loss-of-function screens.",
      "task_counts": {
        "basis": "The source names the complete library but does not print its original total gRNA inventory.",
        "reporting_status": "not_reported",
        "subsets": [],
        "total": null
      },
      "task_formats": [
        "unclassified"
      ],
      "verification": {
        "last_verified": "2026-08-02",
        "notes": "New family admitted after creator source and official resource verification; the unresolved benchmark license is visibly flagged.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": null,
          "evidence_ids": [
            "benchmark-human-crispr-cas9-library-automated-count-evidence",
            "benchmark-human-crispr-cas9-library-automated-version-evidence"
          ],
          "formal_tracks": [],
          "id": "benchmark-human-crispr-cas9-library-initial-release-version",
          "label": "initial-release",
          "notes": "Atlas snapshot label for the creator-paper release; the source does not report a formal benchmark version.",
          "release_date": "2025-02-26",
          "status": "current",
          "task_counts": {
            "basis": "The source names the complete library but does not print its original total gRNA inventory.",
            "reporting_status": "not_reported",
            "subsets": [],
            "total": null
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The pinned repository exposes the Stage-3 workbook and evaluator, but not the complete 21-task evaluation splits or published model outputs. Its Stage-3 workbook has 8,002 non-empty data rows, while the paper describes 8,000 final AI-polished examples.",
        "biosafety_notes": "The suite includes antibody-antigen and RNA-protein interaction prediction. BioBench Atlas stores only metadata and aggregate results and does not mirror sequences.",
        "grader": "The official evaluator is public, including numeric/label extraction, task-specific metrics, and a sentiment-model fallback for some textual classifications.",
        "level": "partially-open",
        "license": null,
        "tasks": "The final paper publishes task definitions, source datasets, split counts, prompts, and metrics; the creator repository does not include the complete evaluation inputs and labels."
      },
      "aliases": [
        "Biology Instructions",
        "Biology-Instructions Evaluation"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "The final paper defines the canonical 21 tasks. Two official-artifact conflicts remain visible: 8,000 versus 8,002 Stage-3 rows, and 21 paper tasks versus 24 evaluator registration keys.",
        "status": "audited-with-caveats",
        "unresolved_fields": 2
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction",
        "classification",
        "regression",
        "scientific-reasoning"
      ],
      "coverage_notes": [
        {
          "count": 5,
          "coverage": "explicitly-in-scope",
          "notes": "Five protein-primary formal tasks; protein-containing interaction tasks are tracked separately.",
          "reporting_status": "reported",
          "tag": "protein-sequence"
        },
        {
          "count": 1,
          "coverage": "explicitly-in-scope",
          "notes": "Antibody-Antigen Neutralization is one formal multi-molecule task.",
          "reporting_status": "reported",
          "tag": "antibody-antigen"
        },
        {
          "count": 1,
          "coverage": "explicitly-in-scope",
          "notes": "RNA-Protein Interaction Prediction is one formal multi-molecule task.",
          "reporting_status": "reported",
          "tag": "rna-protein-binding"
        },
        {
          "count": 0,
          "coverage": "not-in-scope",
          "notes": "Section 3.2.1 explicitly leaves generative applications such as sequence design for future work; property prediction is not relabeled as design.",
          "reporting_status": "reported",
          "tag": "protein-design"
        }
      ],
      "domains": [
        "genomics",
        "transcriptomics",
        "epigenomics",
        "protein-sequence",
        "protein-protein-binding",
        "antibody-antigen",
        "rna-protein-binding",
        "multiomics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-evidence-paper",
          "locator": {
            "note": "Canonical name, affiliations, 21-task taxonomy, explicit exclusion of sequence design, split counts, metrics, prompts, model results, and 8,000-example Stage-3 claim.",
            "type": "table",
            "value": "pp. 17984, 17987-17989, 17992, 17997-18009; Figure 2; Tables 2 and 4-9"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/kind",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/resources",
            "/implementations",
            "/versions/1/release_date",
            "/versions/1/as_of",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets",
            "/versions/1/formal_tracks",
            "/scientific_task_classification/entries/0",
            "/scientific_task_classification/entries/1",
            "/scientific_task_classification/entries/2",
            "/scientific_task_classification/entries/3",
            "/scientific_task_classification/entries/4",
            "/scientific_task_classification/entries/5",
            "/scientific_task_classification/entries/6",
            "/scientific_task_classification/entries/7",
            "/scientific_task_classification/entries/8",
            "/scientific_task_classification/entries/9",
            "/scientific_task_classification/entries/10",
            "/scientific_task_classification/entries/11",
            "/scientific_task_classification/entries/12",
            "/scientific_task_classification/entries/13",
            "/scientific_task_classification/entries/14",
            "/scientific_task_classification/entries/15",
            "/scientific_task_classification/entries/16",
            "/scientific_task_classification/entries/17"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-evidence-preprint",
          "locator": {
            "note": "Initial public date and preprint title.",
            "type": "section",
            "value": "arXiv metadata for 2412.19191 v1"
          },
          "source_id": "bioinstruction-preprint-resource",
          "source_type": "resource",
          "supports": [
            "/release_date",
            "/versions/0/release_date",
            "/versions/0/as_of",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/versions/0/formal_tracks"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-evidence-repository",
          "locator": {
            "note": "Partial release, absent repository license, 24 evaluator registration keys, prompt-independent grader implementation, and public code pin.",
            "type": "repository-path",
            "value": "README.md, evaluation/register_tasks.json, evaluation/evaluate.py at 600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "source_id": "bioinstruction-repository-resource",
          "source_type": "resource",
          "supports": [
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/implementations"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-evidence-stage3",
          "locator": {
            "note": "One worksheet, used range A1:F8003, and 8,002 non-empty data rows; workbook read without modification.",
            "type": "repository-path",
            "value": "stage3.xlsx at 600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "source_id": "bioinstruction-stage3-resource",
          "source_type": "resource",
          "supports": [
            "/access/artifacts",
            "/resources"
          ]
        }
      ],
      "field_status": [
        {
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-evidence-paper",
            "bioinstruction-evidence-stage3"
          ],
          "path": "/access/artifacts",
          "reason": "The paper reports 8,000 final AI-polished Stage-3 examples; the pinned official stage3.xlsx contains 8,002 non-empty data rows.",
          "status": "conflicted"
        },
        {
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-evidence-paper",
            "bioinstruction-evidence-repository"
          ],
          "path": "/implementations",
          "reason": "The final paper defines 21 formal tasks, while the pinned evaluator register_tasks.json contains 24 keys including duplicated or obsolete registrations.",
          "status": "conflicted"
        }
      ],
      "id": "bioinstruction",
      "implementations": [
        {
          "commit": "600acaa08c0302e8f5ce86de0fe041f21c13b53e",
          "framework": "Biology-Instructions official evaluator",
          "notes": "Public post-processing and scoring code; its 24 registration keys must not be treated as the final formal-task count.",
          "status": "official",
          "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e/evaluation"
        }
      ],
      "kind": "suite",
      "latest_version": "emnlp-2025",
      "modalities": [
        "text",
        "dna-rna-sequence",
        "protein-sequence"
      ],
      "name": "Biology-Instructions",
      "organizations": [
        "Shanghai Artificial Intelligence Laboratory",
        "University of Science and Technology of China",
        "University of Sydney",
        "University of Toronto",
        "Chinese University of Hong Kong",
        "Shanghai Jiao Tong University",
        "Fudan University",
        "Shanghai Innovation Institute"
      ],
      "parent_id": null,
      "release_date": "2024-12-26",
      "resources": [
        {
          "access_notes": "Final peer-reviewed creator paper; SHA256 3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a.",
          "id": "bioinstruction-final-paper-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-4.0",
          "pin": {
            "kind": "snapshot",
            "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf",
            "value": "sha256:3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a"
          },
          "type": "paper",
          "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf"
        },
        {
          "access_notes": "Initial public preprint record.",
          "id": "bioinstruction-preprint-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": null,
          "type": "paper",
          "url": "https://arxiv.org/abs/2412.19191"
        },
        {
          "access_notes": "Official repository has no LICENSE file at the pinned commit. register_tasks.json contains 24 keys, including duplicate or obsolete names, rather than the final paper taxonomy of 21 tasks.",
          "id": "bioinstruction-repository-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e",
            "value": "600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "type": "repository",
          "url": "https://github.com/hhnqqq/Biology-Instructions"
        },
        {
          "access_notes": "Pinned workbook SHA256 116ecfe14e55b4a746aa83ae350a6da76635d2892ef2d35d188d1eeb00b60b47; 8,002 non-empty rows across 17 released Stage-3 task labels.",
          "id": "bioinstruction-stage3-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://github.com/hhnqqq/Biology-Instructions/blob/600acaa08c0302e8f5ce86de0fe041f21c13b53e/stage3.xlsx",
            "value": "600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "type": "dataset",
          "url": "https://github.com/hhnqqq/Biology-Instructions/blob/600acaa08c0302e8f5ce86de0fe041f21c13b53e/stage3.xlsx"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "emnlp-2025",
        "entries": [
          {
            "confidence": "high",
            "count": 1,
            "count_basis": "Formal evaluation tracks (Antibody-Antigen Neutralization).",
            "count_ref": null,
            "count_unit": "tracks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Track count; not an example count.",
            "reporting_status": "reported",
            "task_type_id": "antibody-antigen-interaction"
          },
          {
            "confidence": "high",
            "count": 2,
            "count_basis": "Formal evaluation tracks (APA Isoform Prediction and Mean Ribosome Loading).",
            "count_ref": null,
            "count_unit": "tracks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Track count; not an example count.",
            "reporting_status": "reported",
            "task_type_id": "rna-processing-translation-prediction"
          },
          {
            "confidence": "high",
            "count": 2,
            "count_basis": "Formal evaluation tracks (Core Promoter Detection and Promoter Detection 300).",
            "count_ref": null,
            "count_unit": "tracks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Track count; not an example count.",
            "reporting_status": "reported",
            "task_type_id": "promoter-detection"
          },
          {
            "confidence": "high",
            "count": 1,
            "count_basis": "Formal evaluation tracks (CRISPR On-Target Prediction).",
            "count_ref": null,
            "count_unit": "tracks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Track count; not an example count.",
            "reporting_status": "reported",
            "task_type_id": "crispr-guide-activity-prediction"
          },
          {
            "confidence": "high",
            "count": 1,
            "count_basis": "Formal evaluation tracks (Enhancer Activity Prediction).",
            "count_ref": null,
            "count_unit": "tracks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Track count; not an example count.",
            "reporting_status": "reported",
            "task_type_id": "enhancer-activity-prediction"
          },
          {
            "confidence": "high",
            "count": 1,
            "count_basis": "Formal evaluation tracks (Enzyme Commission Number Prediction).",
            "count_ref": null,
            "count_unit": "tracks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Track count; not an example count.",
            "reporting_status": "reported",
            "task_type_id": "protein-function-annotation"
          },
          {
            "confidence": "high",
            "count": 1,
            "count_basis": "Formal evaluation tracks (Epigenetic Marks Prediction).",
            "count_ref": null,
            "count_unit": "tracks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Track count; not an example count.",
            "reporting_status": "reported",
            "task_type_id": "epigenetic-mark-prediction"
          },
          {
            "confidence": "high",
            "count": 1,
            "count_basis": "Formal evaluation tracks (Enhancer-Promoter Interaction Prediction).",
            "count_ref": null,
            "count_unit": "tracks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Track count; not an example count.",
            "reporting_status": "reported",
            "task_type_id": "enhancer-promoter-interaction-prediction"
          },
          {
            "confidence": "high",
            "count": 1,
            "count_basis": "Formal evaluation tracks (Protein Fluorescence Prediction).",
            "count_ref": null,
            "count_unit": "tracks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Track count; not an example count.",
            "reporting_status": "reported",
            "task_type_id": "protein-fluorescence-prediction"
          },
          {
            "confidence": "high",
            "count": 1,
            "count_basis": "Formal evaluation tracks (RNA Modification Prediction).",
            "count_ref": null,
            "count_unit": "tracks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Track count; not an example count.",
            "reporting_status": "reported",
            "task_type_id": "rna-modification-prediction"
          },
          {
            "confidence": "high",
            "count": 1,
            "count_basis": "Formal evaluation tracks (Non-coding RNA Function Classification).",
            "count_ref": null,
            "count_unit": "tracks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Track count; not an example count.",
            "reporting_status": "reported",
            "task_type_id": "rna-function-classification"
          },
          {
            "confidence": "high",
            "count": 1,
            "count_basis": "Formal evaluation tracks (Programmable RNA Switches).",
            "count_ref": null,
            "count_unit": "tracks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Track count; not an example count.",
            "reporting_status": "reported",
            "task_type_id": "rna-design"
          },
          {
            "confidence": "high",
            "count": 1,
            "count_basis": "Formal evaluation tracks (RNA-Protein Interaction Prediction).",
            "count_ref": null,
            "count_unit": "tracks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Track count; not an example count.",
            "reporting_status": "reported",
            "task_type_id": "rna-protein-interaction-prediction"
          },
          {
            "confidence": "high",
            "count": 1,
            "count_basis": "Formal evaluation tracks (siRNA Efficiency Prediction).",
            "count_ref": null,
            "count_unit": "tracks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Track count; not an example count.",
            "reporting_status": "reported",
            "task_type_id": "sirna-efficacy-prediction"
          },
          {
            "confidence": "high",
            "count": 1,
            "count_basis": "Formal evaluation tracks (Protein Solubility Prediction).",
            "count_ref": null,
            "count_unit": "tracks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Track count; not an example count.",
            "reporting_status": "reported",
            "task_type_id": "protein-solubility-prediction"
          },
          {
            "confidence": "high",
            "count": 2,
            "count_basis": "Formal evaluation tracks (Protein Stability and Protein Thermostability).",
            "count_ref": null,
            "count_unit": "tracks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Track count; not an example count.",
            "reporting_status": "reported",
            "task_type_id": "protein-stability-prediction"
          },
          {
            "confidence": "high",
            "count": 2,
            "count_basis": "Formal evaluation tracks (Human and mouse Transcription Binding Sites Detection).",
            "count_ref": null,
            "count_unit": "tracks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Track count; not an example count.",
            "reporting_status": "reported",
            "task_type_id": "transcription-factor-binding-site-prediction"
          },
          {
            "confidence": "high",
            "count": 0,
            "count_basis": "Formal evaluation tracks in the final creator paper.",
            "count_ref": "/coverage_notes/3/count",
            "count_unit": "tracks",
            "coverage": "not-in-scope",
            "evidence_ids": [
              "bioinstruction-evidence-paper"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "The paper explicitly leaves generative sequence design for future work.",
            "reporting_status": "reported",
            "task_type_id": "protein-sequence-design"
          }
        ],
        "notes": "The classification is exhaustive over the 21 formal evaluation tracks in the final creator paper; example counts remain on child records.",
        "status": "complete"
      },
      "summary": "A multi-omics sequence-instruction suite with 21 formal predictive tasks across DNA, RNA, protein, and multi-molecule inputs; the final paper reports task-specific held-out evaluations rather than a single aggregate score.",
      "task_counts": {
        "basis": "formal evaluation tasks in the final creator paper; sample counts are tracked on child records and never added to this task count",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "formal tasks",
            "count": 6,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-dna-tasks",
            "label": "DNA tasks",
            "notes": "EMP, EA, PD300, CPD, TB-H, and TB-M.",
            "reporting_status": "reported"
          },
          {
            "basis": "formal tasks",
            "count": 6,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-rna-tasks",
            "label": "RNA tasks",
            "notes": "APA, ncRNA, Modif, MRL, PRS, and CRI-On.",
            "reporting_status": "reported"
          },
          {
            "basis": "formal tasks",
            "count": 5,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-protein-tasks",
            "label": "Protein tasks",
            "notes": "EC, Stability, Fluorescence, Solubility, and Thermostability.",
            "reporting_status": "reported"
          },
          {
            "basis": "formal tasks",
            "count": 4,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-multi-molecule-tasks",
            "label": "Multi-molecule tasks",
            "notes": "AAN, RPI, EPI, and siRNA.",
            "reporting_status": "reported"
          }
        ],
        "total": 21
      },
      "task_formats": [
        "sequence-conditioned classification and regression instructions"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Canonical name, task taxonomy, row-level split counts, release artifacts, and evaluator were checked against the final paper and pinned official repository.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2024-12-26",
          "evidence_ids": [
            "bioinstruction-evidence-preprint"
          ],
          "formal_tracks": [
            "bioinstruction-emp",
            "bioinstruction-ea",
            "bioinstruction-pd300",
            "bioinstruction-cpd",
            "bioinstruction-tb-human",
            "bioinstruction-tb-mouse",
            "bioinstruction-apa",
            "bioinstruction-ncrna",
            "bioinstruction-modification",
            "bioinstruction-mrl",
            "bioinstruction-prs",
            "bioinstruction-crispr-on-target",
            "bioinstruction-ec",
            "bioinstruction-stability",
            "bioinstruction-fluorescence",
            "bioinstruction-solubility",
            "bioinstruction-thermostability",
            "bioinstruction-aan",
            "bioinstruction-rpi",
            "bioinstruction-epi",
            "bioinstruction-sirna"
          ],
          "id": "bioinstruction-preprint-v1",
          "label": "preprint-v1",
          "notes": "Initial creator preprint.",
          "release_date": "2024-12-26",
          "status": "superseded",
          "task_counts": {
            "basis": "formal evaluation tasks in the final creator paper; sample counts are tracked on child records and never added to this task count",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "formal tasks",
                "count": 6,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-dna-tasks",
                "label": "DNA tasks",
                "notes": "EMP, EA, PD300, CPD, TB-H, and TB-M.",
                "reporting_status": "reported"
              },
              {
                "basis": "formal tasks",
                "count": 6,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-rna-tasks",
                "label": "RNA tasks",
                "notes": "APA, ncRNA, Modif, MRL, PRS, and CRI-On.",
                "reporting_status": "reported"
              },
              {
                "basis": "formal tasks",
                "count": 5,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-protein-tasks",
                "label": "Protein tasks",
                "notes": "EC, Stability, Fluorescence, Solubility, and Thermostability.",
                "reporting_status": "reported"
              },
              {
                "basis": "formal tasks",
                "count": 4,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-multi-molecule-tasks",
                "label": "Multi-molecule tasks",
                "notes": "AAN, RPI, EPI, and siRNA.",
                "reporting_status": "reported"
              }
            ],
            "total": 21
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "bioinstruction-evidence-paper"
          ],
          "formal_tracks": [
            "bioinstruction-emp",
            "bioinstruction-ea",
            "bioinstruction-pd300",
            "bioinstruction-cpd",
            "bioinstruction-tb-human",
            "bioinstruction-tb-mouse",
            "bioinstruction-apa",
            "bioinstruction-ncrna",
            "bioinstruction-modification",
            "bioinstruction-mrl",
            "bioinstruction-prs",
            "bioinstruction-crispr-on-target",
            "bioinstruction-ec",
            "bioinstruction-stability",
            "bioinstruction-fluorescence",
            "bioinstruction-solubility",
            "bioinstruction-thermostability",
            "bioinstruction-aan",
            "bioinstruction-rpi",
            "bioinstruction-epi",
            "bioinstruction-sirna"
          ],
          "id": "bioinstruction-emnlp-2025",
          "label": "emnlp-2025",
          "notes": "Final EMNLP 2025 Findings paper. Table 2 prints 244,681 total test examples, while its 21 task rows sum to 243,227; child records preserve the row-level counts without silently forcing the printed aggregate.",
          "release_date": "2025-11-04",
          "status": "current",
          "task_counts": {
            "basis": "formal evaluation tasks in the final creator paper; sample counts are tracked on child records and never added to this task count",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "formal tasks",
                "count": 6,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-dna-tasks",
                "label": "DNA tasks",
                "notes": "EMP, EA, PD300, CPD, TB-H, and TB-M.",
                "reporting_status": "reported"
              },
              {
                "basis": "formal tasks",
                "count": 6,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-rna-tasks",
                "label": "RNA tasks",
                "notes": "APA, ncRNA, Modif, MRL, PRS, and CRI-On.",
                "reporting_status": "reported"
              },
              {
                "basis": "formal tasks",
                "count": 5,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-protein-tasks",
                "label": "Protein tasks",
                "notes": "EC, Stability, Fluorescence, Solubility, and Thermostability.",
                "reporting_status": "reported"
              },
              {
                "basis": "formal tasks",
                "count": 4,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-multi-molecule-tasks",
                "label": "Multi-molecule tasks",
                "notes": "AAN, RPI, EPI, and siRNA.",
                "reporting_status": "reported"
              }
            ],
            "total": 21
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official post-processing evaluator is public, but the published test examples, labels, and model outputs are not included in the official repository.",
        "biosafety_notes": "BioBench Atlas stores task metadata and aggregate results only and does not mirror the upstream biological sequences.",
        "grader": "keyword/sentiment-assisted binary-label parser followed by Matthews correlation coefficient.",
        "level": "partially-open",
        "license": null,
        "tasks": "The final paper publishes this task's source, definition, and split counts; the creator repository does not include the complete evaluation inputs and labels."
      },
      "aliases": [
        "AAN"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Row-level split counts, task type, metric, prompt, and reported results were checked against the final creator paper.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction",
        "classification"
      ],
      "coverage_notes": [],
      "domains": [
        "protein-sequence",
        "protein-protein-binding",
        "antibody-antigen"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-aan-closed-baselines",
        "bioinstruction-aan-creator-systems",
        "bioinstruction-aan-open-baselines"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-aan-evidence-paper",
          "locator": {
            "note": "Task definition, split counts, input/output format, metric, and creator evaluation.",
            "type": "table",
            "value": "Table 2 (AAN row); Appendix A.2-A.3; Table 8; Tables 4-7"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/parent_id",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/resources",
            "/implementations",
            "/versions/0/release_date",
            "/versions/0/as_of",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-aan-evidence-repository",
          "locator": {
            "note": "Public grader implementation, partial artifact release, and absent repository license.",
            "type": "repository-path",
            "value": "evaluation/evaluate.py and evaluation/register_tasks.json at 600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "source_id": "bioinstruction-aan-repository-resource",
          "source_type": "resource",
          "supports": [
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/implementations"
          ]
        }
      ],
      "field_status": [],
      "id": "bioinstruction-aan",
      "implementations": [
        {
          "commit": "600acaa08c0302e8f5ce86de0fe041f21c13b53e",
          "framework": "Biology-Instructions official evaluator",
          "notes": "Task-specific AAN output parsing and scoring.",
          "status": "official",
          "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e/evaluation"
        }
      ],
      "kind": "track",
      "latest_version": "emnlp-2025",
      "modalities": [
        "text",
        "protein-sequence"
      ],
      "name": "Biology-Instructions Antibody-Antigen Neutralization",
      "organizations": [
        "Shanghai Artificial Intelligence Laboratory",
        "University of Science and Technology of China",
        "University of Sydney",
        "University of Toronto",
        "Chinese University of Hong Kong",
        "Shanghai Jiao Tong University",
        "Fudan University",
        "Shanghai Innovation Institute"
      ],
      "parent_id": "bioinstruction",
      "release_date": "2024-12-26",
      "resources": [
        {
          "access_notes": "Creator paper Table 2 and AAN task/result sections; PDF SHA256 3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a.",
          "id": "bioinstruction-aan-paper-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-4.0",
          "pin": {
            "kind": "snapshot",
            "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf",
            "value": "sha256:3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a"
          },
          "type": "paper",
          "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf"
        },
        {
          "access_notes": "Official evaluator repository; no LICENSE file and no complete evaluation split at the pinned commit.",
          "id": "bioinstruction-aan-repository-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e",
            "value": "600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "type": "repository",
          "url": "https://github.com/hhnqqq/Biology-Instructions"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "emnlp-2025",
        "entries": [
          {
            "confidence": "high",
            "count": 26902,
            "count_basis": "distinct examples across the published train, validation, and test splits",
            "count_ref": "/task_counts/total",
            "count_unit": "examples",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-aan-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Antibody-antigen neutralization prediction.",
            "reporting_status": "reported",
            "task_type_id": "antibody-antigen-interaction"
          }
        ],
        "notes": "Single-purpose formal Biology-Instructions evaluation track.",
        "status": "complete"
      },
      "summary": "Binary neutralization prediction for an antibody-antigen protein-sequence pair, evaluated with Matthews correlation coefficient.",
      "task_counts": {
        "basis": "distinct examples across the published train, validation, and test splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "examples",
            "count": 22359,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-aan-train",
            "label": "Training split",
            "notes": "Published training split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 1242,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-aan-validation",
            "label": "Validation split",
            "notes": "Published validation split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 3301,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-aan-test",
            "label": "Test split",
            "notes": "Held-out split used for creator-paper Tables 4-7.",
            "reporting_status": "reported"
          }
        ],
        "total": 26902
      },
      "task_formats": [
        "paired protein-sequence binary classification"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task-level audit complete.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "bioinstruction-aan-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "bioinstruction-aan-emnlp-2025",
          "label": "emnlp-2025",
          "notes": "Final paper split: 22359 train / 1242 validation / 3301 test examples.",
          "release_date": "2025-11-04",
          "status": "current",
          "task_counts": {
            "basis": "distinct examples across the published train, validation, and test splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "examples",
                "count": 22359,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-aan-train",
                "label": "Training split",
                "notes": "Published training split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 1242,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-aan-validation",
                "label": "Validation split",
                "notes": "Published validation split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 3301,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-aan-test",
                "label": "Test split",
                "notes": "Held-out split used for creator-paper Tables 4-7.",
                "reporting_status": "reported"
              }
            ],
            "total": 26902
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official post-processing evaluator is public, but the published test examples, labels, and model outputs are not included in the official repository.",
        "biosafety_notes": "BioBench Atlas stores task metadata and aggregate results only and does not mirror the upstream biological sequences.",
        "grader": "first-number parser followed by squared Pearson correlation labeled R2.",
        "level": "partially-open",
        "license": null,
        "tasks": "The final paper publishes this task's source, definition, and split counts; the creator repository does not include the complete evaluation inputs and labels."
      },
      "aliases": [
        "APA"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Row-level split counts, task type, metric, prompt, and reported results were checked against the final creator paper.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction",
        "regression"
      ],
      "coverage_notes": [],
      "domains": [
        "transcriptomics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-apa-closed-baselines",
        "bioinstruction-apa-creator-systems",
        "bioinstruction-apa-open-baselines"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-apa-evidence-paper",
          "locator": {
            "note": "Task definition, split counts, input/output format, metric, and creator evaluation.",
            "type": "table",
            "value": "Table 2 (APA row); Appendix A.2-A.3; Table 8; Tables 4-7"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/parent_id",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/resources",
            "/implementations",
            "/versions/0/release_date",
            "/versions/0/as_of",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-apa-evidence-repository",
          "locator": {
            "note": "Public grader implementation, partial artifact release, and absent repository license.",
            "type": "repository-path",
            "value": "evaluation/evaluate.py and evaluation/register_tasks.json at 600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "source_id": "bioinstruction-apa-repository-resource",
          "source_type": "resource",
          "supports": [
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/implementations"
          ]
        }
      ],
      "field_status": [],
      "id": "bioinstruction-apa",
      "implementations": [
        {
          "commit": "600acaa08c0302e8f5ce86de0fe041f21c13b53e",
          "framework": "Biology-Instructions official evaluator",
          "notes": "Task-specific APA output parsing and scoring.",
          "status": "official",
          "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e/evaluation"
        }
      ],
      "kind": "track",
      "latest_version": "emnlp-2025",
      "modalities": [
        "text",
        "dna-rna-sequence"
      ],
      "name": "Biology-Instructions APA Isoform Prediction",
      "organizations": [
        "Shanghai Artificial Intelligence Laboratory",
        "University of Science and Technology of China",
        "University of Sydney",
        "University of Toronto",
        "Chinese University of Hong Kong",
        "Shanghai Jiao Tong University",
        "Fudan University",
        "Shanghai Innovation Institute"
      ],
      "parent_id": "bioinstruction",
      "release_date": "2024-12-26",
      "resources": [
        {
          "access_notes": "Creator paper Table 2 and APA task/result sections; PDF SHA256 3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a.",
          "id": "bioinstruction-apa-paper-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-4.0",
          "pin": {
            "kind": "snapshot",
            "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf",
            "value": "sha256:3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a"
          },
          "type": "paper",
          "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf"
        },
        {
          "access_notes": "Official evaluator repository; no LICENSE file and no complete evaluation split at the pinned commit.",
          "id": "bioinstruction-apa-repository-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e",
            "value": "600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "type": "repository",
          "url": "https://github.com/hhnqqq/Biology-Instructions"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "emnlp-2025",
        "entries": [
          {
            "confidence": "high",
            "count": 1658482,
            "count_basis": "distinct examples across the published train, validation, and test splits",
            "count_ref": "/task_counts/total",
            "count_unit": "examples",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-apa-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Alternative-polyadenylation isoform usage prediction.",
            "reporting_status": "reported",
            "task_type_id": "rna-processing-translation-prediction"
          }
        ],
        "notes": "Single-purpose formal Biology-Instructions evaluation track.",
        "status": "complete"
      },
      "summary": "Regression of alternative-polyadenylation isoform usage from an RNA sequence, reported as squared Pearson correlation under the paper's R2 label.",
      "task_counts": {
        "basis": "distinct examples across the published train, validation, and test splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "examples",
            "count": 1575557,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-apa-train",
            "label": "Training split",
            "notes": "Published training split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 33170,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-apa-validation",
            "label": "Validation split",
            "notes": "Published validation split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 49755,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-apa-test",
            "label": "Test split",
            "notes": "Held-out split used for creator-paper Tables 4-7.",
            "reporting_status": "reported"
          }
        ],
        "total": 1658482
      },
      "task_formats": [
        "RNA-sequence regression"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task-level audit complete.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "bioinstruction-apa-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "bioinstruction-apa-emnlp-2025",
          "label": "emnlp-2025",
          "notes": "Final paper split: 1575557 train / 33170 validation / 49755 test examples.",
          "release_date": "2025-11-04",
          "status": "current",
          "task_counts": {
            "basis": "distinct examples across the published train, validation, and test splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "examples",
                "count": 1575557,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-apa-train",
                "label": "Training split",
                "notes": "Published training split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 33170,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-apa-validation",
                "label": "Validation split",
                "notes": "Published validation split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 49755,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-apa-test",
                "label": "Test split",
                "notes": "Held-out split used for creator-paper Tables 4-7.",
                "reporting_status": "reported"
              }
            ],
            "total": 1658482
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official post-processing evaluator is public, but the published test examples, labels, and model outputs are not included in the official repository.",
        "biosafety_notes": "BioBench Atlas stores task metadata and aggregate results only and does not mirror the upstream biological sequences.",
        "grader": "keyword/sentiment-assisted binary-label parser followed by Matthews correlation coefficient.",
        "level": "partially-open",
        "license": null,
        "tasks": "The final paper publishes this task's source, definition, and split counts; the creator repository does not include the complete evaluation inputs and labels."
      },
      "aliases": [
        "CPD"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Row-level split counts, task type, metric, prompt, and reported results were checked against the final creator paper.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction",
        "classification"
      ],
      "coverage_notes": [],
      "domains": [
        "genomics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-cpd-closed-baselines",
        "bioinstruction-cpd-creator-systems",
        "bioinstruction-cpd-open-baselines"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-cpd-evidence-paper",
          "locator": {
            "note": "Task definition, split counts, input/output format, metric, and creator evaluation.",
            "type": "table",
            "value": "Table 2 (CPD row); Appendix A.2-A.3; Table 8; Tables 4-7"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/parent_id",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/resources",
            "/implementations",
            "/versions/0/release_date",
            "/versions/0/as_of",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-cpd-evidence-repository",
          "locator": {
            "note": "Public grader implementation, partial artifact release, and absent repository license.",
            "type": "repository-path",
            "value": "evaluation/evaluate.py and evaluation/register_tasks.json at 600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "source_id": "bioinstruction-cpd-repository-resource",
          "source_type": "resource",
          "supports": [
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/implementations"
          ]
        }
      ],
      "field_status": [],
      "id": "bioinstruction-cpd",
      "implementations": [
        {
          "commit": "600acaa08c0302e8f5ce86de0fe041f21c13b53e",
          "framework": "Biology-Instructions official evaluator",
          "notes": "Task-specific CPD output parsing and scoring.",
          "status": "official",
          "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e/evaluation"
        }
      ],
      "kind": "track",
      "latest_version": "emnlp-2025",
      "modalities": [
        "text",
        "dna-rna-sequence"
      ],
      "name": "Biology-Instructions Core Promoter Detection",
      "organizations": [
        "Shanghai Artificial Intelligence Laboratory",
        "University of Science and Technology of China",
        "University of Sydney",
        "University of Toronto",
        "Chinese University of Hong Kong",
        "Shanghai Jiao Tong University",
        "Fudan University",
        "Shanghai Innovation Institute"
      ],
      "parent_id": "bioinstruction",
      "release_date": "2024-12-26",
      "resources": [
        {
          "access_notes": "Creator paper Table 2 and CPD task/result sections; PDF SHA256 3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a.",
          "id": "bioinstruction-cpd-paper-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-4.0",
          "pin": {
            "kind": "snapshot",
            "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf",
            "value": "sha256:3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a"
          },
          "type": "paper",
          "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf"
        },
        {
          "access_notes": "Official evaluator repository; no LICENSE file and no complete evaluation split at the pinned commit.",
          "id": "bioinstruction-cpd-repository-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e",
            "value": "600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "type": "repository",
          "url": "https://github.com/hhnqqq/Biology-Instructions"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "emnlp-2025",
        "entries": [
          {
            "confidence": "high",
            "count": 118392,
            "count_basis": "distinct examples across the published train, validation, and test splits",
            "count_ref": "/task_counts/total",
            "count_unit": "examples",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-cpd-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Core-promoter detection.",
            "reporting_status": "reported",
            "task_type_id": "promoter-detection"
          }
        ],
        "notes": "Single-purpose formal Biology-Instructions evaluation track.",
        "status": "complete"
      },
      "summary": "Binary detection of a core promoter in a short DNA sequence, evaluated with Matthews correlation coefficient.",
      "task_counts": {
        "basis": "distinct examples across the published train, validation, and test splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "examples",
            "count": 94712,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-cpd-train",
            "label": "Training split",
            "notes": "Published training split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 11840,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-cpd-validation",
            "label": "Validation split",
            "notes": "Published validation split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 11840,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-cpd-test",
            "label": "Test split",
            "notes": "Held-out split used for creator-paper Tables 4-7.",
            "reporting_status": "reported"
          }
        ],
        "total": 118392
      },
      "task_formats": [
        "DNA-sequence binary classification"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task-level audit complete.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "bioinstruction-cpd-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "bioinstruction-cpd-emnlp-2025",
          "label": "emnlp-2025",
          "notes": "Final paper split: 94712 train / 11840 validation / 11840 test examples.",
          "release_date": "2025-11-04",
          "status": "current",
          "task_counts": {
            "basis": "distinct examples across the published train, validation, and test splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "examples",
                "count": 94712,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-cpd-train",
                "label": "Training split",
                "notes": "Published training split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 11840,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-cpd-validation",
                "label": "Validation split",
                "notes": "Published validation split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 11840,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-cpd-test",
                "label": "Test split",
                "notes": "Held-out split used for creator-paper Tables 4-7.",
                "reporting_status": "reported"
              }
            ],
            "total": 118392
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official post-processing evaluator is public, but the published test examples, labels, and model outputs are not included in the official repository.",
        "biosafety_notes": "BioBench Atlas stores task metadata and aggregate results only and does not mirror the upstream biological sequences.",
        "grader": "first-number parser followed by Spearman rank correlation.",
        "level": "partially-open",
        "license": null,
        "tasks": "The final paper publishes this task's source, definition, and split counts; the creator repository does not include the complete evaluation inputs and labels."
      },
      "aliases": [
        "CRI-On"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Row-level split counts, task type, metric, prompt, and reported results were checked against the final creator paper.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction",
        "regression"
      ],
      "coverage_notes": [],
      "domains": [
        "genomics",
        "transcriptomics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-crispr-on-target-closed-baselines",
        "bioinstruction-crispr-on-target-creator-systems",
        "bioinstruction-crispr-on-target-open-baselines"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-crispr-on-target-evidence-paper",
          "locator": {
            "note": "Task definition, split counts, input/output format, metric, and creator evaluation.",
            "type": "table",
            "value": "Table 2 (CRI-On row); Appendix A.2-A.3; Table 8; Tables 4-7"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/parent_id",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/resources",
            "/implementations",
            "/versions/0/release_date",
            "/versions/0/as_of",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-crispr-on-target-evidence-repository",
          "locator": {
            "note": "Public grader implementation, partial artifact release, and absent repository license.",
            "type": "repository-path",
            "value": "evaluation/evaluate.py and evaluation/register_tasks.json at 600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "source_id": "bioinstruction-crispr-on-target-repository-resource",
          "source_type": "resource",
          "supports": [
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/implementations"
          ]
        }
      ],
      "field_status": [],
      "id": "bioinstruction-crispr-on-target",
      "implementations": [
        {
          "commit": "600acaa08c0302e8f5ce86de0fe041f21c13b53e",
          "framework": "Biology-Instructions official evaluator",
          "notes": "Task-specific CRI-On output parsing and scoring.",
          "status": "official",
          "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e/evaluation"
        }
      ],
      "kind": "track",
      "latest_version": "emnlp-2025",
      "modalities": [
        "text",
        "dna-rna-sequence"
      ],
      "name": "Biology-Instructions CRISPR On-Target Prediction",
      "organizations": [
        "Shanghai Artificial Intelligence Laboratory",
        "University of Science and Technology of China",
        "University of Sydney",
        "University of Toronto",
        "Chinese University of Hong Kong",
        "Shanghai Jiao Tong University",
        "Fudan University",
        "Shanghai Innovation Institute"
      ],
      "parent_id": "bioinstruction",
      "release_date": "2024-12-26",
      "resources": [
        {
          "access_notes": "Creator paper Table 2 and CRI-On task/result sections; PDF SHA256 3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a.",
          "id": "bioinstruction-crispr-on-target-paper-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-4.0",
          "pin": {
            "kind": "snapshot",
            "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf",
            "value": "sha256:3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a"
          },
          "type": "paper",
          "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf"
        },
        {
          "access_notes": "Official evaluator repository; no LICENSE file and no complete evaluation split at the pinned commit.",
          "id": "bioinstruction-crispr-on-target-repository-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e",
            "value": "600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "type": "repository",
          "url": "https://github.com/hhnqqq/Biology-Instructions"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "emnlp-2025",
        "entries": [
          {
            "confidence": "high",
            "count": 2076,
            "count_basis": "distinct examples across the published train, validation, and test splits",
            "count_ref": "/task_counts/total",
            "count_unit": "examples",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-crispr-on-target-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "CRISPR guide on-target activity prediction.",
            "reporting_status": "reported",
            "task_type_id": "crispr-guide-activity-prediction"
          }
        ],
        "notes": "Single-purpose formal Biology-Instructions evaluation track.",
        "status": "complete"
      },
      "summary": "Regression of CRISPR guide on-target activity from an RNA sequence, evaluated with Spearman rank correlation.",
      "task_counts": {
        "basis": "distinct examples across the published train, validation, and test splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "examples",
            "count": 1453,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-crispr-on-target-train",
            "label": "Training split",
            "notes": "Published training split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 207,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-crispr-on-target-validation",
            "label": "Validation split",
            "notes": "Published validation split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 416,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-crispr-on-target-test",
            "label": "Test split",
            "notes": "Held-out split used for creator-paper Tables 4-7.",
            "reporting_status": "reported"
          }
        ],
        "total": 2076
      },
      "task_formats": [
        "guide-RNA sequence regression"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task-level audit complete.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "bioinstruction-crispr-on-target-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "bioinstruction-crispr-on-target-emnlp-2025",
          "label": "emnlp-2025",
          "notes": "Final paper split: 1453 train / 207 validation / 416 test examples.",
          "release_date": "2025-11-04",
          "status": "current",
          "task_counts": {
            "basis": "distinct examples across the published train, validation, and test splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "examples",
                "count": 1453,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-crispr-on-target-train",
                "label": "Training split",
                "notes": "Published training split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 207,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-crispr-on-target-validation",
                "label": "Validation split",
                "notes": "Published validation split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 416,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-crispr-on-target-test",
                "label": "Test split",
                "notes": "Held-out split used for creator-paper Tables 4-7.",
                "reporting_status": "reported"
              }
            ],
            "total": 2076
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official post-processing evaluator is public, but the published test examples, labels, and model outputs are not included in the official repository.",
        "biosafety_notes": "BioBench Atlas stores task metadata and aggregate results only and does not mirror the upstream biological sequences.",
        "grader": "two-number parser followed by separate housekeeping and developmental Pearson correlations.",
        "level": "partially-open",
        "license": null,
        "tasks": "The final paper publishes this task's source, definition, and split counts; the creator repository does not include the complete evaluation inputs and labels."
      },
      "aliases": [
        "EA"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Row-level split counts, task type, metric, prompt, and reported results were checked against the final creator paper.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction",
        "regression"
      ],
      "coverage_notes": [],
      "domains": [
        "genomics",
        "epigenomics",
        "assay-screening"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-ea-closed-baselines",
        "bioinstruction-ea-creator-systems",
        "bioinstruction-ea-open-baselines"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-ea-evidence-paper",
          "locator": {
            "note": "Task definition, split counts, input/output format, metric, and creator evaluation.",
            "type": "table",
            "value": "Table 2 (EA row); Appendix A.2-A.3; Table 8; Tables 4-7"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/parent_id",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/resources",
            "/implementations",
            "/versions/0/release_date",
            "/versions/0/as_of",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-ea-evidence-repository",
          "locator": {
            "note": "Public grader implementation, partial artifact release, and absent repository license.",
            "type": "repository-path",
            "value": "evaluation/evaluate.py and evaluation/register_tasks.json at 600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "source_id": "bioinstruction-ea-repository-resource",
          "source_type": "resource",
          "supports": [
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/implementations"
          ]
        }
      ],
      "field_status": [],
      "id": "bioinstruction-ea",
      "implementations": [
        {
          "commit": "600acaa08c0302e8f5ce86de0fe041f21c13b53e",
          "framework": "Biology-Instructions official evaluator",
          "notes": "Task-specific EA output parsing and scoring.",
          "status": "official",
          "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e/evaluation"
        }
      ],
      "kind": "track",
      "latest_version": "emnlp-2025",
      "modalities": [
        "text",
        "dna-rna-sequence"
      ],
      "name": "Biology-Instructions Enhancer Activity Prediction",
      "organizations": [
        "Shanghai Artificial Intelligence Laboratory",
        "University of Science and Technology of China",
        "University of Sydney",
        "University of Toronto",
        "Chinese University of Hong Kong",
        "Shanghai Jiao Tong University",
        "Fudan University",
        "Shanghai Innovation Institute"
      ],
      "parent_id": "bioinstruction",
      "release_date": "2024-12-26",
      "resources": [
        {
          "access_notes": "Creator paper Table 2 and EA task/result sections; PDF SHA256 3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a.",
          "id": "bioinstruction-ea-paper-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-4.0",
          "pin": {
            "kind": "snapshot",
            "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf",
            "value": "sha256:3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a"
          },
          "type": "paper",
          "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf"
        },
        {
          "access_notes": "Official evaluator repository; no LICENSE file and no complete evaluation split at the pinned commit.",
          "id": "bioinstruction-ea-repository-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e",
            "value": "600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "type": "repository",
          "url": "https://github.com/hhnqqq/Biology-Instructions"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "emnlp-2025",
        "entries": [
          {
            "confidence": "high",
            "count": 484052,
            "count_basis": "distinct examples across the published train, validation, and test splits",
            "count_ref": "/task_counts/total",
            "count_unit": "examples",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-ea-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Enhancer activity prediction.",
            "reporting_status": "reported",
            "task_type_id": "enhancer-activity-prediction"
          }
        ],
        "notes": "Single-purpose formal Biology-Instructions evaluation track.",
        "status": "complete"
      },
      "summary": "Two-output regression of housekeeping and developmental enhancer activity from a DNA sequence, evaluated with separate Pearson correlations.",
      "task_counts": {
        "basis": "distinct examples across the published train, validation, and test splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "examples",
            "count": 402296,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-ea-train",
            "label": "Training split",
            "notes": "Published training split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 40570,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-ea-validation",
            "label": "Validation split",
            "notes": "Published validation split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 41186,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-ea-test",
            "label": "Test split",
            "notes": "Held-out split used for creator-paper Tables 4-7.",
            "reporting_status": "reported"
          }
        ],
        "total": 484052
      },
      "task_formats": [
        "DNA-sequence two-output regression"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task-level audit complete.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "bioinstruction-ea-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "bioinstruction-ea-emnlp-2025",
          "label": "emnlp-2025",
          "notes": "Final paper split: 402296 train / 40570 validation / 41186 test examples.",
          "release_date": "2025-11-04",
          "status": "current",
          "task_counts": {
            "basis": "distinct examples across the published train, validation, and test splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "examples",
                "count": 402296,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-ea-train",
                "label": "Training split",
                "notes": "Published training split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 40570,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-ea-validation",
                "label": "Validation split",
                "notes": "Published validation split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 41186,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-ea-test",
                "label": "Test split",
                "notes": "Held-out split used for creator-paper Tables 4-7.",
                "reporting_status": "reported"
              }
            ],
            "total": 484052
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official post-processing evaluator is public, but the published test examples, labels, and model outputs are not included in the official repository.",
        "biosafety_notes": "BioBench Atlas stores task metadata and aggregate results only and does not mirror the upstream biological sequences.",
        "grader": "EC-number regex and multi-hot conversion followed by creator Fmax implementation.",
        "level": "partially-open",
        "license": null,
        "tasks": "The final paper publishes this task's source, definition, and split counts; the creator repository does not include the complete evaluation inputs and labels."
      },
      "aliases": [
        "EC"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Row-level split counts, task type, metric, prompt, and reported results were checked against the final creator paper.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction",
        "classification"
      ],
      "coverage_notes": [],
      "domains": [
        "protein-sequence",
        "proteomics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-ec-closed-baselines",
        "bioinstruction-ec-creator-systems",
        "bioinstruction-ec-open-baselines"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-ec-evidence-paper",
          "locator": {
            "note": "Task definition, split counts, input/output format, metric, and creator evaluation.",
            "type": "table",
            "value": "Table 2 (EC row); Appendix A.2-A.3; Table 8; Tables 4-7"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/parent_id",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/resources",
            "/implementations",
            "/versions/0/release_date",
            "/versions/0/as_of",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-ec-evidence-repository",
          "locator": {
            "note": "Public grader implementation, partial artifact release, and absent repository license.",
            "type": "repository-path",
            "value": "evaluation/evaluate.py and evaluation/register_tasks.json at 600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "source_id": "bioinstruction-ec-repository-resource",
          "source_type": "resource",
          "supports": [
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/implementations"
          ]
        }
      ],
      "field_status": [],
      "id": "bioinstruction-ec",
      "implementations": [
        {
          "commit": "600acaa08c0302e8f5ce86de0fe041f21c13b53e",
          "framework": "Biology-Instructions official evaluator",
          "notes": "Task-specific EC output parsing and scoring.",
          "status": "official",
          "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e/evaluation"
        }
      ],
      "kind": "track",
      "latest_version": "emnlp-2025",
      "modalities": [
        "text",
        "protein-sequence"
      ],
      "name": "Biology-Instructions Enzyme Commission Number Prediction",
      "organizations": [
        "Shanghai Artificial Intelligence Laboratory",
        "University of Science and Technology of China",
        "University of Sydney",
        "University of Toronto",
        "Chinese University of Hong Kong",
        "Shanghai Jiao Tong University",
        "Fudan University",
        "Shanghai Innovation Institute"
      ],
      "parent_id": "bioinstruction",
      "release_date": "2024-12-26",
      "resources": [
        {
          "access_notes": "Creator paper Table 2 and EC task/result sections; PDF SHA256 3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a.",
          "id": "bioinstruction-ec-paper-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-4.0",
          "pin": {
            "kind": "snapshot",
            "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf",
            "value": "sha256:3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a"
          },
          "type": "paper",
          "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf"
        },
        {
          "access_notes": "Official evaluator repository; no LICENSE file and no complete evaluation split at the pinned commit.",
          "id": "bioinstruction-ec-repository-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e",
            "value": "600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "type": "repository",
          "url": "https://github.com/hhnqqq/Biology-Instructions"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "emnlp-2025",
        "entries": [
          {
            "confidence": "high",
            "count": 19199,
            "count_basis": "distinct examples across the published train, validation, and test splits",
            "count_ref": "/task_counts/total",
            "count_unit": "examples",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-ec-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Enzyme Commission function annotation.",
            "reporting_status": "reported",
            "task_type_id": "protein-function-annotation"
          }
        ],
        "notes": "Single-purpose formal Biology-Instructions evaluation track.",
        "status": "complete"
      },
      "summary": "Multi-label Enzyme Commission number prediction from a protein sequence, evaluated with the creator's Fmax implementation.",
      "task_counts": {
        "basis": "distinct examples across the published train, validation, and test splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "examples",
            "count": 15551,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-ec-train",
            "label": "Training split",
            "notes": "Published training split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 1729,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-ec-validation",
            "label": "Validation split",
            "notes": "Published validation split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 1919,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-ec-test",
            "label": "Test split",
            "notes": "Held-out split used for creator-paper Tables 4-7.",
            "reporting_status": "reported"
          }
        ],
        "total": 19199
      },
      "task_formats": [
        "protein-sequence multi-label classification"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task-level audit complete.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "bioinstruction-ec-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "bioinstruction-ec-emnlp-2025",
          "label": "emnlp-2025",
          "notes": "Final paper split: 15551 train / 1729 validation / 1919 test examples.",
          "release_date": "2025-11-04",
          "status": "current",
          "task_counts": {
            "basis": "distinct examples across the published train, validation, and test splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "examples",
                "count": 15551,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-ec-train",
                "label": "Training split",
                "notes": "Published training split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 1729,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-ec-validation",
                "label": "Validation split",
                "notes": "Published validation split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 1919,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-ec-test",
                "label": "Test split",
                "notes": "Held-out split used for creator-paper Tables 4-7.",
                "reporting_status": "reported"
              }
            ],
            "total": 19199
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official post-processing evaluator is public, but the published test examples, labels, and model outputs are not included in the official repository.",
        "biosafety_notes": "BioBench Atlas stores task metadata and aggregate results only and does not mirror the upstream biological sequences.",
        "grader": "keyword/sentiment-assisted binary-label parser followed by Matthews correlation coefficient.",
        "level": "partially-open",
        "license": null,
        "tasks": "The final paper publishes this task's source, definition, and split counts; the creator repository does not include the complete evaluation inputs and labels."
      },
      "aliases": [
        "EMP"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Row-level split counts, task type, metric, prompt, and reported results were checked against the final creator paper.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction",
        "classification"
      ],
      "coverage_notes": [],
      "domains": [
        "genomics",
        "epigenomics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-emp-closed-baselines",
        "bioinstruction-emp-creator-systems",
        "bioinstruction-emp-open-baselines"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-emp-evidence-paper",
          "locator": {
            "note": "Task definition, split counts, input/output format, metric, and creator evaluation.",
            "type": "table",
            "value": "Table 2 (EMP row); Appendix A.2-A.3; Table 8; Tables 4-7"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/parent_id",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/resources",
            "/implementations",
            "/versions/0/release_date",
            "/versions/0/as_of",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-emp-evidence-repository",
          "locator": {
            "note": "Public grader implementation, partial artifact release, and absent repository license.",
            "type": "repository-path",
            "value": "evaluation/evaluate.py and evaluation/register_tasks.json at 600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "source_id": "bioinstruction-emp-repository-resource",
          "source_type": "resource",
          "supports": [
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/implementations"
          ]
        }
      ],
      "field_status": [],
      "id": "bioinstruction-emp",
      "implementations": [
        {
          "commit": "600acaa08c0302e8f5ce86de0fe041f21c13b53e",
          "framework": "Biology-Instructions official evaluator",
          "notes": "Task-specific EMP output parsing and scoring.",
          "status": "official",
          "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e/evaluation"
        }
      ],
      "kind": "track",
      "latest_version": "emnlp-2025",
      "modalities": [
        "text",
        "dna-rna-sequence"
      ],
      "name": "Biology-Instructions Epigenetic Marks Prediction",
      "organizations": [
        "Shanghai Artificial Intelligence Laboratory",
        "University of Science and Technology of China",
        "University of Sydney",
        "University of Toronto",
        "Chinese University of Hong Kong",
        "Shanghai Jiao Tong University",
        "Fudan University",
        "Shanghai Innovation Institute"
      ],
      "parent_id": "bioinstruction",
      "release_date": "2024-12-26",
      "resources": [
        {
          "access_notes": "Creator paper Table 2 and EMP task/result sections; PDF SHA256 3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a.",
          "id": "bioinstruction-emp-paper-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-4.0",
          "pin": {
            "kind": "snapshot",
            "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf",
            "value": "sha256:3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a"
          },
          "type": "paper",
          "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf"
        },
        {
          "access_notes": "Official evaluator repository; no LICENSE file and no complete evaluation split at the pinned commit.",
          "id": "bioinstruction-emp-repository-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e",
            "value": "600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "type": "repository",
          "url": "https://github.com/hhnqqq/Biology-Instructions"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "emnlp-2025",
        "entries": [
          {
            "confidence": "high",
            "count": 287367,
            "count_basis": "distinct examples across the published train, validation, and test splits",
            "count_ref": "/task_counts/total",
            "count_unit": "examples",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-emp-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Epigenetic-mark prediction.",
            "reporting_status": "reported",
            "task_type_id": "epigenetic-mark-prediction"
          }
        ],
        "notes": "Single-purpose formal Biology-Instructions evaluation track.",
        "status": "complete"
      },
      "summary": "Binary prediction of whether a DNA sequence carries an epigenetic mark, evaluated with Matthews correlation coefficient.",
      "task_counts": {
        "basis": "distinct examples across the published train, validation, and test splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "examples",
            "count": 229885,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-emp-train",
            "label": "Training split",
            "notes": "Published training split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 28741,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-emp-validation",
            "label": "Validation split",
            "notes": "Published validation split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 28741,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-emp-test",
            "label": "Test split",
            "notes": "Held-out split used for creator-paper Tables 4-7.",
            "reporting_status": "reported"
          }
        ],
        "total": 287367
      },
      "task_formats": [
        "DNA-sequence binary classification"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task-level audit complete.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "bioinstruction-emp-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "bioinstruction-emp-emnlp-2025",
          "label": "emnlp-2025",
          "notes": "Final paper split: 229885 train / 28741 validation / 28741 test examples.",
          "release_date": "2025-11-04",
          "status": "current",
          "task_counts": {
            "basis": "distinct examples across the published train, validation, and test splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "examples",
                "count": 229885,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-emp-train",
                "label": "Training split",
                "notes": "Published training split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 28741,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-emp-validation",
                "label": "Validation split",
                "notes": "Published validation split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 28741,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-emp-test",
                "label": "Test split",
                "notes": "Held-out split used for creator-paper Tables 4-7.",
                "reporting_status": "reported"
              }
            ],
            "total": 287367
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official post-processing evaluator is public, but the published test examples, labels, and model outputs are not included in the official repository.",
        "biosafety_notes": "BioBench Atlas stores task metadata and aggregate results only and does not mirror the upstream biological sequences.",
        "grader": "keyword/sentiment-assisted binary-label parser followed by Matthews correlation coefficient.",
        "level": "partially-open",
        "license": null,
        "tasks": "The final paper publishes this task's source, definition, and split counts; the creator repository does not include the complete evaluation inputs and labels."
      },
      "aliases": [
        "EPI"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Row-level split counts, task type, metric, prompt, and reported results were checked against the final creator paper.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction",
        "classification"
      ],
      "coverage_notes": [],
      "domains": [
        "genomics",
        "epigenomics",
        "molecular-cell-biology"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-epi-closed-baselines",
        "bioinstruction-epi-creator-systems",
        "bioinstruction-epi-open-baselines"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-epi-evidence-paper",
          "locator": {
            "note": "Task definition, split counts, input/output format, metric, and creator evaluation.",
            "type": "table",
            "value": "Table 2 (EPI row); Appendix A.2-A.3; Table 8; Tables 4-7"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/parent_id",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/resources",
            "/implementations",
            "/versions/0/release_date",
            "/versions/0/as_of",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-epi-evidence-repository",
          "locator": {
            "note": "Public grader implementation, partial artifact release, and absent repository license.",
            "type": "repository-path",
            "value": "evaluation/evaluate.py and evaluation/register_tasks.json at 600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "source_id": "bioinstruction-epi-repository-resource",
          "source_type": "resource",
          "supports": [
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/implementations"
          ]
        }
      ],
      "field_status": [],
      "id": "bioinstruction-epi",
      "implementations": [
        {
          "commit": "600acaa08c0302e8f5ce86de0fe041f21c13b53e",
          "framework": "Biology-Instructions official evaluator",
          "notes": "Task-specific EPI output parsing and scoring.",
          "status": "official",
          "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e/evaluation"
        }
      ],
      "kind": "track",
      "latest_version": "emnlp-2025",
      "modalities": [
        "text",
        "dna-rna-sequence"
      ],
      "name": "Biology-Instructions Enhancer-Promoter Interaction Prediction",
      "organizations": [
        "Shanghai Artificial Intelligence Laboratory",
        "University of Science and Technology of China",
        "University of Sydney",
        "University of Toronto",
        "Chinese University of Hong Kong",
        "Shanghai Jiao Tong University",
        "Fudan University",
        "Shanghai Innovation Institute"
      ],
      "parent_id": "bioinstruction",
      "release_date": "2024-12-26",
      "resources": [
        {
          "access_notes": "Creator paper Table 2 and EPI task/result sections; PDF SHA256 3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a.",
          "id": "bioinstruction-epi-paper-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-4.0",
          "pin": {
            "kind": "snapshot",
            "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf",
            "value": "sha256:3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a"
          },
          "type": "paper",
          "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf"
        },
        {
          "access_notes": "Official evaluator repository; no LICENSE file and no complete evaluation split at the pinned commit.",
          "id": "bioinstruction-epi-repository-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e",
            "value": "600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "type": "repository",
          "url": "https://github.com/hhnqqq/Biology-Instructions"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "emnlp-2025",
        "entries": [
          {
            "confidence": "high",
            "count": 16368,
            "count_basis": "distinct examples across the published train, validation, and test splits",
            "count_ref": "/task_counts/total",
            "count_unit": "examples",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-epi-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Enhancer-promoter interaction prediction.",
            "reporting_status": "reported",
            "task_type_id": "enhancer-promoter-interaction-prediction"
          }
        ],
        "notes": "Single-purpose formal Biology-Instructions evaluation track.",
        "status": "complete"
      },
      "summary": "Binary interaction prediction for enhancer and promoter DNA sequences, evaluated with Matthews correlation coefficient.",
      "task_counts": {
        "basis": "distinct examples across the published train, validation, and test splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "examples",
            "count": 14288,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-epi-train",
            "label": "Training split",
            "notes": "Published training split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 1772,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-epi-validation",
            "label": "Validation split",
            "notes": "Published validation split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 308,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-epi-test",
            "label": "Test split",
            "notes": "Held-out split used for creator-paper Tables 4-7.",
            "reporting_status": "reported"
          }
        ],
        "total": 16368
      },
      "task_formats": [
        "paired DNA-sequence binary classification"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task-level audit complete.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "bioinstruction-epi-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "bioinstruction-epi-emnlp-2025",
          "label": "emnlp-2025",
          "notes": "Final paper split: 14288 train / 1772 validation / 308 test examples.",
          "release_date": "2025-11-04",
          "status": "current",
          "task_counts": {
            "basis": "distinct examples across the published train, validation, and test splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "examples",
                "count": 14288,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-epi-train",
                "label": "Training split",
                "notes": "Published training split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 1772,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-epi-validation",
                "label": "Validation split",
                "notes": "Published validation split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 308,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-epi-test",
                "label": "Test split",
                "notes": "Held-out split used for creator-paper Tables 4-7.",
                "reporting_status": "reported"
              }
            ],
            "total": 16368
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official post-processing evaluator is public, but the published test examples, labels, and model outputs are not included in the official repository.",
        "biosafety_notes": "BioBench Atlas stores task metadata and aggregate results only and does not mirror the upstream biological sequences.",
        "grader": "first-number parser followed by Spearman rank correlation.",
        "level": "partially-open",
        "license": null,
        "tasks": "The final paper publishes this task's source, definition, and split counts; the creator repository does not include the complete evaluation inputs and labels."
      },
      "aliases": [
        "Flu"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Row-level split counts, task type, metric, prompt, and reported results were checked against the final creator paper.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction",
        "regression"
      ],
      "coverage_notes": [],
      "domains": [
        "protein-sequence",
        "assay-screening"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-fluorescence-closed-baselines",
        "bioinstruction-fluorescence-creator-systems",
        "bioinstruction-fluorescence-open-baselines"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-fluorescence-evidence-paper",
          "locator": {
            "note": "Task definition, split counts, input/output format, metric, and creator evaluation.",
            "type": "table",
            "value": "Table 2 (Flu row); Appendix A.2-A.3; Table 8; Tables 4-7"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/parent_id",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/resources",
            "/implementations",
            "/versions/0/release_date",
            "/versions/0/as_of",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-fluorescence-evidence-repository",
          "locator": {
            "note": "Public grader implementation, partial artifact release, and absent repository license.",
            "type": "repository-path",
            "value": "evaluation/evaluate.py and evaluation/register_tasks.json at 600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "source_id": "bioinstruction-fluorescence-repository-resource",
          "source_type": "resource",
          "supports": [
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/implementations"
          ]
        }
      ],
      "field_status": [],
      "id": "bioinstruction-fluorescence",
      "implementations": [
        {
          "commit": "600acaa08c0302e8f5ce86de0fe041f21c13b53e",
          "framework": "Biology-Instructions official evaluator",
          "notes": "Task-specific Flu output parsing and scoring.",
          "status": "official",
          "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e/evaluation"
        }
      ],
      "kind": "track",
      "latest_version": "emnlp-2025",
      "modalities": [
        "text",
        "protein-sequence"
      ],
      "name": "Biology-Instructions Protein Fluorescence Prediction",
      "organizations": [
        "Shanghai Artificial Intelligence Laboratory",
        "University of Science and Technology of China",
        "University of Sydney",
        "University of Toronto",
        "Chinese University of Hong Kong",
        "Shanghai Jiao Tong University",
        "Fudan University",
        "Shanghai Innovation Institute"
      ],
      "parent_id": "bioinstruction",
      "release_date": "2024-12-26",
      "resources": [
        {
          "access_notes": "Creator paper Table 2 and Flu task/result sections; PDF SHA256 3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a.",
          "id": "bioinstruction-fluorescence-paper-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-4.0",
          "pin": {
            "kind": "snapshot",
            "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf",
            "value": "sha256:3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a"
          },
          "type": "paper",
          "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf"
        },
        {
          "access_notes": "Official evaluator repository; no LICENSE file and no complete evaluation split at the pinned commit.",
          "id": "bioinstruction-fluorescence-repository-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e",
            "value": "600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "type": "repository",
          "url": "https://github.com/hhnqqq/Biology-Instructions"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "emnlp-2025",
        "entries": [
          {
            "confidence": "high",
            "count": 54025,
            "count_basis": "distinct examples across the published train, validation, and test splits",
            "count_ref": "/task_counts/total",
            "count_unit": "examples",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-fluorescence-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Protein fluorescence regression.",
            "reporting_status": "reported",
            "task_type_id": "protein-fluorescence-prediction"
          }
        ],
        "notes": "Single-purpose formal Biology-Instructions evaluation track.",
        "status": "complete"
      },
      "summary": "Regression of protein fluorescence from an amino-acid sequence, evaluated with Spearman rank correlation.",
      "task_counts": {
        "basis": "distinct examples across the published train, validation, and test splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "examples",
            "count": 21446,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-fluorescence-train",
            "label": "Training split",
            "notes": "Published training split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 5362,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-fluorescence-validation",
            "label": "Validation split",
            "notes": "Published validation split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 27217,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-fluorescence-test",
            "label": "Test split",
            "notes": "Held-out split used for creator-paper Tables 4-7.",
            "reporting_status": "reported"
          }
        ],
        "total": 54025
      },
      "task_formats": [
        "protein-sequence regression"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task-level audit complete.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "bioinstruction-fluorescence-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "bioinstruction-fluorescence-emnlp-2025",
          "label": "emnlp-2025",
          "notes": "Final paper split: 21446 train / 5362 validation / 27217 test examples.",
          "release_date": "2025-11-04",
          "status": "current",
          "task_counts": {
            "basis": "distinct examples across the published train, validation, and test splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "examples",
                "count": 21446,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-fluorescence-train",
                "label": "Training split",
                "notes": "Published training split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 5362,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-fluorescence-validation",
                "label": "Validation split",
                "notes": "Published validation split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 27217,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-fluorescence-test",
                "label": "Test split",
                "notes": "Held-out split used for creator-paper Tables 4-7.",
                "reporting_status": "reported"
              }
            ],
            "total": 54025
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official post-processing evaluator is public, but the published test examples, labels, and model outputs are not included in the official repository.",
        "biosafety_notes": "BioBench Atlas stores task metadata and aggregate results only and does not mirror the upstream biological sequences.",
        "grader": "modification-label extractor with sentiment fallback for none, followed by macro ROC AUC.",
        "level": "partially-open",
        "license": null,
        "tasks": "The final paper publishes this task's source, definition, and split counts; the creator repository does not include the complete evaluation inputs and labels."
      },
      "aliases": [
        "Modif"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Row-level split counts, task type, metric, prompt, and reported results were checked against the final creator paper.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction",
        "classification"
      ],
      "coverage_notes": [],
      "domains": [
        "transcriptomics",
        "epigenomics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-modification-closed-baselines",
        "bioinstruction-modification-creator-systems",
        "bioinstruction-modification-open-baselines"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-modification-evidence-paper",
          "locator": {
            "note": "Task definition, split counts, input/output format, metric, and creator evaluation.",
            "type": "table",
            "value": "Table 2 (Modif row); Appendix A.2-A.3; Table 8; Tables 4-7"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/parent_id",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/resources",
            "/implementations",
            "/versions/0/release_date",
            "/versions/0/as_of",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-modification-evidence-repository",
          "locator": {
            "note": "Public grader implementation, partial artifact release, and absent repository license.",
            "type": "repository-path",
            "value": "evaluation/evaluate.py and evaluation/register_tasks.json at 600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "source_id": "bioinstruction-modification-repository-resource",
          "source_type": "resource",
          "supports": [
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/implementations"
          ]
        }
      ],
      "field_status": [],
      "id": "bioinstruction-modification",
      "implementations": [
        {
          "commit": "600acaa08c0302e8f5ce86de0fe041f21c13b53e",
          "framework": "Biology-Instructions official evaluator",
          "notes": "Task-specific Modif output parsing and scoring.",
          "status": "official",
          "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e/evaluation"
        }
      ],
      "kind": "track",
      "latest_version": "emnlp-2025",
      "modalities": [
        "text",
        "dna-rna-sequence"
      ],
      "name": "Biology-Instructions RNA Modification Prediction",
      "organizations": [
        "Shanghai Artificial Intelligence Laboratory",
        "University of Science and Technology of China",
        "University of Sydney",
        "University of Toronto",
        "Chinese University of Hong Kong",
        "Shanghai Jiao Tong University",
        "Fudan University",
        "Shanghai Innovation Institute"
      ],
      "parent_id": "bioinstruction",
      "release_date": "2024-12-26",
      "resources": [
        {
          "access_notes": "Creator paper Table 2 and Modif task/result sections; PDF SHA256 3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a.",
          "id": "bioinstruction-modification-paper-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-4.0",
          "pin": {
            "kind": "snapshot",
            "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf",
            "value": "sha256:3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a"
          },
          "type": "paper",
          "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf"
        },
        {
          "access_notes": "Official evaluator repository; no LICENSE file and no complete evaluation split at the pinned commit.",
          "id": "bioinstruction-modification-repository-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e",
            "value": "600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "type": "repository",
          "url": "https://github.com/hhnqqq/Biology-Instructions"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "emnlp-2025",
        "entries": [
          {
            "confidence": "high",
            "count": 309460,
            "count_basis": "distinct examples across the published train, validation, and test splits",
            "count_ref": "/task_counts/total",
            "count_unit": "examples",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-modification-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "RNA chemical-modification prediction.",
            "reporting_status": "reported",
            "task_type_id": "rna-modification-prediction"
          }
        ],
        "notes": "Single-purpose formal Biology-Instructions evaluation track.",
        "status": "complete"
      },
      "summary": "Multi-label prediction of RNA chemical modifications, evaluated with macro area under the ROC curve.",
      "task_counts": {
        "basis": "distinct examples across the published train, validation, and test splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "examples",
            "count": 304661,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-modification-train",
            "label": "Training split",
            "notes": "Published training split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 3599,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-modification-validation",
            "label": "Validation split",
            "notes": "Published validation split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 1200,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-modification-test",
            "label": "Test split",
            "notes": "Held-out split used for creator-paper Tables 4-7.",
            "reporting_status": "reported"
          }
        ],
        "total": 309460
      },
      "task_formats": [
        "RNA-sequence multi-label classification"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task-level audit complete.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "bioinstruction-modification-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "bioinstruction-modification-emnlp-2025",
          "label": "emnlp-2025",
          "notes": "Final paper split: 304661 train / 3599 validation / 1200 test examples.",
          "release_date": "2025-11-04",
          "status": "current",
          "task_counts": {
            "basis": "distinct examples across the published train, validation, and test splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "examples",
                "count": 304661,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-modification-train",
                "label": "Training split",
                "notes": "Published training split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 3599,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-modification-validation",
                "label": "Validation split",
                "notes": "Published validation split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 1200,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-modification-test",
                "label": "Test split",
                "notes": "Held-out split used for creator-paper Tables 4-7.",
                "reporting_status": "reported"
              }
            ],
            "total": 309460
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official post-processing evaluator is public, but the published test examples, labels, and model outputs are not included in the official repository.",
        "biosafety_notes": "BioBench Atlas stores task metadata and aggregate results only and does not mirror the upstream biological sequences.",
        "grader": "first-number parser followed by squared Pearson correlation labeled R2.",
        "level": "partially-open",
        "license": null,
        "tasks": "The final paper publishes this task's source, definition, and split counts; the creator repository does not include the complete evaluation inputs and labels."
      },
      "aliases": [
        "MRL"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Row-level split counts, task type, metric, prompt, and reported results were checked against the final creator paper.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction",
        "regression"
      ],
      "coverage_notes": [],
      "domains": [
        "transcriptomics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-mrl-closed-baselines",
        "bioinstruction-mrl-creator-systems",
        "bioinstruction-mrl-open-baselines"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-mrl-evidence-paper",
          "locator": {
            "note": "Task definition, split counts, input/output format, metric, and creator evaluation.",
            "type": "table",
            "value": "Table 2 (MRL row); Appendix A.2-A.3; Table 8; Tables 4-7"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/parent_id",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/resources",
            "/implementations",
            "/versions/0/release_date",
            "/versions/0/as_of",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-mrl-evidence-repository",
          "locator": {
            "note": "Public grader implementation, partial artifact release, and absent repository license.",
            "type": "repository-path",
            "value": "evaluation/evaluate.py and evaluation/register_tasks.json at 600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "source_id": "bioinstruction-mrl-repository-resource",
          "source_type": "resource",
          "supports": [
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/implementations"
          ]
        }
      ],
      "field_status": [],
      "id": "bioinstruction-mrl",
      "implementations": [
        {
          "commit": "600acaa08c0302e8f5ce86de0fe041f21c13b53e",
          "framework": "Biology-Instructions official evaluator",
          "notes": "Task-specific MRL output parsing and scoring.",
          "status": "official",
          "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e/evaluation"
        }
      ],
      "kind": "track",
      "latest_version": "emnlp-2025",
      "modalities": [
        "text",
        "dna-rna-sequence"
      ],
      "name": "Biology-Instructions Mean Ribosome Loading Prediction",
      "organizations": [
        "Shanghai Artificial Intelligence Laboratory",
        "University of Science and Technology of China",
        "University of Sydney",
        "University of Toronto",
        "Chinese University of Hong Kong",
        "Shanghai Jiao Tong University",
        "Fudan University",
        "Shanghai Innovation Institute"
      ],
      "parent_id": "bioinstruction",
      "release_date": "2024-12-26",
      "resources": [
        {
          "access_notes": "Creator paper Table 2 and MRL task/result sections; PDF SHA256 3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a.",
          "id": "bioinstruction-mrl-paper-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-4.0",
          "pin": {
            "kind": "snapshot",
            "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf",
            "value": "sha256:3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a"
          },
          "type": "paper",
          "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf"
        },
        {
          "access_notes": "Official evaluator repository; no LICENSE file and no complete evaluation split at the pinned commit.",
          "id": "bioinstruction-mrl-repository-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e",
            "value": "600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "type": "repository",
          "url": "https://github.com/hhnqqq/Biology-Instructions"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "emnlp-2025",
        "entries": [
          {
            "confidence": "high",
            "count": 91519,
            "count_basis": "distinct examples across the published train, validation, and test splits",
            "count_ref": "/task_counts/total",
            "count_unit": "examples",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-mrl-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Mean ribosome-loading prediction.",
            "reporting_status": "reported",
            "task_type_id": "rna-processing-translation-prediction"
          }
        ],
        "notes": "Single-purpose formal Biology-Instructions evaluation track.",
        "status": "complete"
      },
      "summary": "Regression of mean ribosome loading from an RNA sequence, reported as squared Pearson correlation under the paper's R2 label.",
      "task_counts": {
        "basis": "distinct examples across the published train, validation, and test splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "examples",
            "count": 76319,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-mrl-train",
            "label": "Training split",
            "notes": "Published training split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 7600,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-mrl-validation",
            "label": "Validation split",
            "notes": "Published validation split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 7600,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-mrl-test",
            "label": "Test split",
            "notes": "Held-out split used for creator-paper Tables 4-7.",
            "reporting_status": "reported"
          }
        ],
        "total": 91519
      },
      "task_formats": [
        "RNA-sequence regression"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task-level audit complete.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "bioinstruction-mrl-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "bioinstruction-mrl-emnlp-2025",
          "label": "emnlp-2025",
          "notes": "Final paper split: 76319 train / 7600 validation / 7600 test examples.",
          "release_date": "2025-11-04",
          "status": "current",
          "task_counts": {
            "basis": "distinct examples across the published train, validation, and test splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "examples",
                "count": 76319,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-mrl-train",
                "label": "Training split",
                "notes": "Published training split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 7600,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-mrl-validation",
                "label": "Validation split",
                "notes": "Published validation split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 7600,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-mrl-test",
                "label": "Test split",
                "notes": "Held-out split used for creator-paper Tables 4-7.",
                "reporting_status": "reported"
              }
            ],
            "total": 91519
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official post-processing evaluator is public, but the published test examples, labels, and model outputs are not included in the official repository.",
        "biosafety_notes": "BioBench Atlas stores task metadata and aggregate results only and does not mirror the upstream biological sequences.",
        "grader": "ordered class-name extractor followed by exact accuracy.",
        "level": "partially-open",
        "license": null,
        "tasks": "The final paper publishes this task's source, definition, and split counts; the creator repository does not include the complete evaluation inputs and labels."
      },
      "aliases": [
        "ncRNA"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Row-level split counts, task type, metric, prompt, and reported results were checked against the final creator paper.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction",
        "classification"
      ],
      "coverage_notes": [],
      "domains": [
        "transcriptomics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-ncrna-closed-baselines",
        "bioinstruction-ncrna-creator-systems",
        "bioinstruction-ncrna-open-baselines"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-ncrna-evidence-paper",
          "locator": {
            "note": "Task definition, split counts, input/output format, metric, and creator evaluation.",
            "type": "table",
            "value": "Table 2 (ncRNA row); Appendix A.2-A.3; Table 8; Tables 4-7"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/parent_id",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/resources",
            "/implementations",
            "/versions/0/release_date",
            "/versions/0/as_of",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-ncrna-evidence-repository",
          "locator": {
            "note": "Public grader implementation, partial artifact release, and absent repository license.",
            "type": "repository-path",
            "value": "evaluation/evaluate.py and evaluation/register_tasks.json at 600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "source_id": "bioinstruction-ncrna-repository-resource",
          "source_type": "resource",
          "supports": [
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/implementations"
          ]
        }
      ],
      "field_status": [],
      "id": "bioinstruction-ncrna",
      "implementations": [
        {
          "commit": "600acaa08c0302e8f5ce86de0fe041f21c13b53e",
          "framework": "Biology-Instructions official evaluator",
          "notes": "Task-specific ncRNA output parsing and scoring.",
          "status": "official",
          "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e/evaluation"
        }
      ],
      "kind": "track",
      "latest_version": "emnlp-2025",
      "modalities": [
        "text",
        "dna-rna-sequence"
      ],
      "name": "Biology-Instructions Non-coding RNA Function Classification",
      "organizations": [
        "Shanghai Artificial Intelligence Laboratory",
        "University of Science and Technology of China",
        "University of Sydney",
        "University of Toronto",
        "Chinese University of Hong Kong",
        "Shanghai Jiao Tong University",
        "Fudan University",
        "Shanghai Innovation Institute"
      ],
      "parent_id": "bioinstruction",
      "release_date": "2024-12-26",
      "resources": [
        {
          "access_notes": "Creator paper Table 2 and ncRNA task/result sections; PDF SHA256 3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a.",
          "id": "bioinstruction-ncrna-paper-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-4.0",
          "pin": {
            "kind": "snapshot",
            "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf",
            "value": "sha256:3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a"
          },
          "type": "paper",
          "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf"
        },
        {
          "access_notes": "Official evaluator repository; no LICENSE file and no complete evaluation split at the pinned commit.",
          "id": "bioinstruction-ncrna-repository-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e",
            "value": "600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "type": "repository",
          "url": "https://github.com/hhnqqq/Biology-Instructions"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "emnlp-2025",
        "entries": [
          {
            "confidence": "high",
            "count": 11160,
            "count_basis": "distinct examples across the published train, validation, and test splits",
            "count_ref": "/task_counts/total",
            "count_unit": "examples",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-ncrna-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Non-coding RNA function classification.",
            "reporting_status": "reported",
            "task_type_id": "rna-function-classification"
          }
        ],
        "notes": "Single-purpose formal Biology-Instructions evaluation track.",
        "status": "complete"
      },
      "summary": "Thirteen-class functional classification of a non-coding RNA sequence, evaluated with exact extracted-label accuracy.",
      "task_counts": {
        "basis": "distinct examples across the published train, validation, and test splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "examples",
            "count": 5670,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-ncrna-train",
            "label": "Training split",
            "notes": "Published training split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 650,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-ncrna-validation",
            "label": "Validation split",
            "notes": "Published validation split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 4840,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-ncrna-test",
            "label": "Test split",
            "notes": "Held-out split used for creator-paper Tables 4-7.",
            "reporting_status": "reported"
          }
        ],
        "total": 11160
      },
      "task_formats": [
        "RNA-sequence 13-class classification"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task-level audit complete.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "bioinstruction-ncrna-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "bioinstruction-ncrna-emnlp-2025",
          "label": "emnlp-2025",
          "notes": "Final paper split: 5670 train / 650 validation / 4840 test examples.",
          "release_date": "2025-11-04",
          "status": "current",
          "task_counts": {
            "basis": "distinct examples across the published train, validation, and test splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "examples",
                "count": 5670,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-ncrna-train",
                "label": "Training split",
                "notes": "Published training split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 650,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-ncrna-validation",
                "label": "Validation split",
                "notes": "Published validation split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 4840,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-ncrna-test",
                "label": "Test split",
                "notes": "Held-out split used for creator-paper Tables 4-7.",
                "reporting_status": "reported"
              }
            ],
            "total": 11160
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official post-processing evaluator is public, but the published test examples, labels, and model outputs are not included in the official repository.",
        "biosafety_notes": "BioBench Atlas stores task metadata and aggregate results only and does not mirror the upstream biological sequences.",
        "grader": "keyword/sentiment-assisted binary-label parser followed by Matthews correlation coefficient.",
        "level": "partially-open",
        "license": null,
        "tasks": "The final paper publishes this task's source, definition, and split counts; the creator repository does not include the complete evaluation inputs and labels."
      },
      "aliases": [
        "PD300"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Row-level split counts, task type, metric, prompt, and reported results were checked against the final creator paper.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction",
        "classification"
      ],
      "coverage_notes": [],
      "domains": [
        "genomics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-pd300-closed-baselines",
        "bioinstruction-pd300-creator-systems",
        "bioinstruction-pd300-open-baselines"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-pd300-evidence-paper",
          "locator": {
            "note": "Task definition, split counts, input/output format, metric, and creator evaluation.",
            "type": "table",
            "value": "Table 2 (PD300 row); Appendix A.2-A.3; Table 8; Tables 4-7"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/parent_id",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/resources",
            "/implementations",
            "/versions/0/release_date",
            "/versions/0/as_of",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-pd300-evidence-repository",
          "locator": {
            "note": "Public grader implementation, partial artifact release, and absent repository license.",
            "type": "repository-path",
            "value": "evaluation/evaluate.py and evaluation/register_tasks.json at 600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "source_id": "bioinstruction-pd300-repository-resource",
          "source_type": "resource",
          "supports": [
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/implementations"
          ]
        }
      ],
      "field_status": [],
      "id": "bioinstruction-pd300",
      "implementations": [
        {
          "commit": "600acaa08c0302e8f5ce86de0fe041f21c13b53e",
          "framework": "Biology-Instructions official evaluator",
          "notes": "Task-specific PD300 output parsing and scoring.",
          "status": "official",
          "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e/evaluation"
        }
      ],
      "kind": "track",
      "latest_version": "emnlp-2025",
      "modalities": [
        "text",
        "dna-rna-sequence"
      ],
      "name": "Biology-Instructions Promoter Detection 300",
      "organizations": [
        "Shanghai Artificial Intelligence Laboratory",
        "University of Science and Technology of China",
        "University of Sydney",
        "University of Toronto",
        "Chinese University of Hong Kong",
        "Shanghai Jiao Tong University",
        "Fudan University",
        "Shanghai Innovation Institute"
      ],
      "parent_id": "bioinstruction",
      "release_date": "2024-12-26",
      "resources": [
        {
          "access_notes": "Creator paper Table 2 and PD300 task/result sections; PDF SHA256 3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a.",
          "id": "bioinstruction-pd300-paper-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-4.0",
          "pin": {
            "kind": "snapshot",
            "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf",
            "value": "sha256:3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a"
          },
          "type": "paper",
          "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf"
        },
        {
          "access_notes": "Official evaluator repository; no LICENSE file and no complete evaluation split at the pinned commit.",
          "id": "bioinstruction-pd300-repository-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e",
            "value": "600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "type": "repository",
          "url": "https://github.com/hhnqqq/Biology-Instructions"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "emnlp-2025",
        "entries": [
          {
            "confidence": "high",
            "count": 118392,
            "count_basis": "distinct examples across the published train, validation, and test splits",
            "count_ref": "/task_counts/total",
            "count_unit": "examples",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-pd300-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Promoter detection in 300-base-pair context.",
            "reporting_status": "reported",
            "task_type_id": "promoter-detection"
          }
        ],
        "notes": "Single-purpose formal Biology-Instructions evaluation track.",
        "status": "complete"
      },
      "summary": "Binary promoter detection in a 300-base-pair DNA context, evaluated with Matthews correlation coefficient.",
      "task_counts": {
        "basis": "distinct examples across the published train, validation, and test splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "examples",
            "count": 94712,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-pd300-train",
            "label": "Training split",
            "notes": "Published training split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 11840,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-pd300-validation",
            "label": "Validation split",
            "notes": "Published validation split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 11840,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-pd300-test",
            "label": "Test split",
            "notes": "Held-out split used for creator-paper Tables 4-7.",
            "reporting_status": "reported"
          }
        ],
        "total": 118392
      },
      "task_formats": [
        "DNA-sequence binary classification"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task-level audit complete.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "bioinstruction-pd300-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "bioinstruction-pd300-emnlp-2025",
          "label": "emnlp-2025",
          "notes": "Final paper split: 94712 train / 11840 validation / 11840 test examples.",
          "release_date": "2025-11-04",
          "status": "current",
          "task_counts": {
            "basis": "distinct examples across the published train, validation, and test splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "examples",
                "count": 94712,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-pd300-train",
                "label": "Training split",
                "notes": "Published training split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 11840,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-pd300-validation",
                "label": "Validation split",
                "notes": "Published validation split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 11840,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-pd300-test",
                "label": "Test split",
                "notes": "Held-out split used for creator-paper Tables 4-7.",
                "reporting_status": "reported"
              }
            ],
            "total": 118392
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official post-processing evaluator is public, but the published test examples, labels, and model outputs are not included in the official repository.",
        "biosafety_notes": "BioBench Atlas stores task metadata and aggregate results only and does not mirror the upstream biological sequences.",
        "grader": "three-number parser followed by mean squared Pearson correlation across ON, OFF, and ON/OFF.",
        "level": "partially-open",
        "license": null,
        "tasks": "The final paper publishes this task's source, definition, and split counts; the creator repository does not include the complete evaluation inputs and labels."
      },
      "aliases": [
        "PRS"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Row-level split counts, task type, metric, prompt, and reported results were checked against the final creator paper.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction",
        "regression"
      ],
      "coverage_notes": [],
      "domains": [
        "transcriptomics",
        "molecular-cell-biology"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-prs-closed-baselines",
        "bioinstruction-prs-creator-systems",
        "bioinstruction-prs-open-baselines"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-prs-evidence-paper",
          "locator": {
            "note": "Task definition, split counts, input/output format, metric, and creator evaluation.",
            "type": "table",
            "value": "Table 2 (PRS row); Appendix A.2-A.3; Table 8; Tables 4-7"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/parent_id",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/resources",
            "/implementations",
            "/versions/0/release_date",
            "/versions/0/as_of",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-prs-evidence-repository",
          "locator": {
            "note": "Public grader implementation, partial artifact release, and absent repository license.",
            "type": "repository-path",
            "value": "evaluation/evaluate.py and evaluation/register_tasks.json at 600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "source_id": "bioinstruction-prs-repository-resource",
          "source_type": "resource",
          "supports": [
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/implementations"
          ]
        }
      ],
      "field_status": [],
      "id": "bioinstruction-prs",
      "implementations": [
        {
          "commit": "600acaa08c0302e8f5ce86de0fe041f21c13b53e",
          "framework": "Biology-Instructions official evaluator",
          "notes": "Task-specific PRS output parsing and scoring.",
          "status": "official",
          "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e/evaluation"
        }
      ],
      "kind": "track",
      "latest_version": "emnlp-2025",
      "modalities": [
        "text",
        "dna-rna-sequence"
      ],
      "name": "Biology-Instructions Programmable RNA Switches",
      "organizations": [
        "Shanghai Artificial Intelligence Laboratory",
        "University of Science and Technology of China",
        "University of Sydney",
        "University of Toronto",
        "Chinese University of Hong Kong",
        "Shanghai Jiao Tong University",
        "Fudan University",
        "Shanghai Innovation Institute"
      ],
      "parent_id": "bioinstruction",
      "release_date": "2024-12-26",
      "resources": [
        {
          "access_notes": "Creator paper Table 2 and PRS task/result sections; PDF SHA256 3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a.",
          "id": "bioinstruction-prs-paper-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-4.0",
          "pin": {
            "kind": "snapshot",
            "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf",
            "value": "sha256:3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a"
          },
          "type": "paper",
          "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf"
        },
        {
          "access_notes": "Official evaluator repository; no LICENSE file and no complete evaluation split at the pinned commit.",
          "id": "bioinstruction-prs-repository-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e",
            "value": "600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "type": "repository",
          "url": "https://github.com/hhnqqq/Biology-Instructions"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "emnlp-2025",
        "entries": [
          {
            "confidence": "high",
            "count": 93399,
            "count_basis": "distinct examples across the published train, validation, and test splits",
            "count_ref": "/task_counts/total",
            "count_unit": "examples",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-prs-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Programmable RNA-switch value prediction.",
            "reporting_status": "reported",
            "task_type_id": "rna-design"
          }
        ],
        "notes": "Single-purpose formal Biology-Instructions evaluation track.",
        "status": "complete"
      },
      "summary": "Three-output regression of ON, OFF, and ON/OFF programmable RNA-switch values, aggregated as their mean squared Pearson correlation.",
      "task_counts": {
        "basis": "distinct examples across the published train, validation, and test splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "examples",
            "count": 73227,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-prs-train",
            "label": "Training split",
            "notes": "Published training split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 9153,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-prs-validation",
            "label": "Validation split",
            "notes": "Published validation split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 11019,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-prs-test",
            "label": "Test split",
            "notes": "Held-out split used for creator-paper Tables 4-7.",
            "reporting_status": "reported"
          }
        ],
        "total": 93399
      },
      "task_formats": [
        "RNA-sequence three-output regression"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task-level audit complete.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "bioinstruction-prs-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "bioinstruction-prs-emnlp-2025",
          "label": "emnlp-2025",
          "notes": "Final paper split: 73227 train / 9153 validation / 11019 test examples.",
          "release_date": "2025-11-04",
          "status": "current",
          "task_counts": {
            "basis": "distinct examples across the published train, validation, and test splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "examples",
                "count": 73227,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-prs-train",
                "label": "Training split",
                "notes": "Published training split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 9153,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-prs-validation",
                "label": "Validation split",
                "notes": "Published validation split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 11019,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-prs-test",
                "label": "Test split",
                "notes": "Held-out split used for creator-paper Tables 4-7.",
                "reporting_status": "reported"
              }
            ],
            "total": 93399
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official post-processing evaluator is public, but the published test examples, labels, and model outputs are not included in the official repository.",
        "biosafety_notes": "BioBench Atlas stores task metadata and aggregate results only and does not mirror the upstream biological sequences.",
        "grader": "keyword/sentiment-assisted binary-label parser followed by Matthews correlation coefficient.",
        "level": "partially-open",
        "license": null,
        "tasks": "The final paper publishes this task's source, definition, and split counts; the creator repository does not include the complete evaluation inputs and labels."
      },
      "aliases": [
        "RPI"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Row-level split counts, task type, metric, prompt, and reported results were checked against the final creator paper.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction",
        "classification"
      ],
      "coverage_notes": [],
      "domains": [
        "transcriptomics",
        "protein-sequence",
        "rna-protein-binding",
        "multiomics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-rpi-closed-baselines",
        "bioinstruction-rpi-creator-systems",
        "bioinstruction-rpi-open-baselines"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-rpi-evidence-paper",
          "locator": {
            "note": "Task definition, split counts, input/output format, metric, and creator evaluation.",
            "type": "table",
            "value": "Table 2 (RPI row); Appendix A.2-A.3; Table 8; Tables 4-7"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/parent_id",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/resources",
            "/implementations",
            "/versions/0/release_date",
            "/versions/0/as_of",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-rpi-evidence-repository",
          "locator": {
            "note": "Public grader implementation, partial artifact release, and absent repository license.",
            "type": "repository-path",
            "value": "evaluation/evaluate.py and evaluation/register_tasks.json at 600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "source_id": "bioinstruction-rpi-repository-resource",
          "source_type": "resource",
          "supports": [
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/implementations"
          ]
        }
      ],
      "field_status": [],
      "id": "bioinstruction-rpi",
      "implementations": [
        {
          "commit": "600acaa08c0302e8f5ce86de0fe041f21c13b53e",
          "framework": "Biology-Instructions official evaluator",
          "notes": "Task-specific RPI output parsing and scoring.",
          "status": "official",
          "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e/evaluation"
        }
      ],
      "kind": "track",
      "latest_version": "emnlp-2025",
      "modalities": [
        "text",
        "dna-rna-sequence",
        "protein-sequence"
      ],
      "name": "Biology-Instructions RNA-Protein Interaction Prediction",
      "organizations": [
        "Shanghai Artificial Intelligence Laboratory",
        "University of Science and Technology of China",
        "University of Sydney",
        "University of Toronto",
        "Chinese University of Hong Kong",
        "Shanghai Jiao Tong University",
        "Fudan University",
        "Shanghai Innovation Institute"
      ],
      "parent_id": "bioinstruction",
      "release_date": "2024-12-26",
      "resources": [
        {
          "access_notes": "Creator paper Table 2 and RPI task/result sections; PDF SHA256 3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a.",
          "id": "bioinstruction-rpi-paper-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-4.0",
          "pin": {
            "kind": "snapshot",
            "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf",
            "value": "sha256:3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a"
          },
          "type": "paper",
          "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf"
        },
        {
          "access_notes": "Official evaluator repository; no LICENSE file and no complete evaluation split at the pinned commit.",
          "id": "bioinstruction-rpi-repository-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e",
            "value": "600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "type": "repository",
          "url": "https://github.com/hhnqqq/Biology-Instructions"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "emnlp-2025",
        "entries": [
          {
            "confidence": "high",
            "count": 20824,
            "count_basis": "distinct examples across the published train, validation, and test splits",
            "count_ref": "/task_counts/total",
            "count_unit": "examples",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-rpi-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "RNA-protein interaction prediction.",
            "reporting_status": "reported",
            "task_type_id": "rna-protein-interaction-prediction"
          }
        ],
        "notes": "Single-purpose formal Biology-Instructions evaluation track.",
        "status": "complete"
      },
      "summary": "Binary interaction prediction for an RNA and protein sequence pair, evaluated with Matthews correlation coefficient.",
      "task_counts": {
        "basis": "distinct examples across the published train, validation, and test splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "examples",
            "count": 14994,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-rpi-train",
            "label": "Training split",
            "notes": "Published training split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 1666,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-rpi-validation",
            "label": "Validation split",
            "notes": "Published validation split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 4164,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-rpi-test",
            "label": "Test split",
            "notes": "Held-out split used for creator-paper Tables 4-7.",
            "reporting_status": "reported"
          }
        ],
        "total": 20824
      },
      "task_formats": [
        "RNA-protein sequence-pair binary classification"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task-level audit complete.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "bioinstruction-rpi-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "bioinstruction-rpi-emnlp-2025",
          "label": "emnlp-2025",
          "notes": "Final paper split: 14994 train / 1666 validation / 4164 test examples.",
          "release_date": "2025-11-04",
          "status": "current",
          "task_counts": {
            "basis": "distinct examples across the published train, validation, and test splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "examples",
                "count": 14994,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-rpi-train",
                "label": "Training split",
                "notes": "Published training split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 1666,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-rpi-validation",
                "label": "Validation split",
                "notes": "Published validation split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 4164,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-rpi-test",
                "label": "Test split",
                "notes": "Held-out split used for creator-paper Tables 4-7.",
                "reporting_status": "reported"
              }
            ],
            "total": 20824
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official post-processing evaluator is public, but the published test examples, labels, and model outputs are not included in the official repository.",
        "biosafety_notes": "BioBench Atlas stores task metadata and aggregate results only and does not mirror the upstream biological sequences.",
        "grader": "first-number parser followed by the creator mixed score combining MAE, range MAE, and binary F1.",
        "level": "partially-open",
        "license": null,
        "tasks": "The final paper publishes this task's source, definition, and split counts; the creator repository does not include the complete evaluation inputs and labels."
      },
      "aliases": [
        "siRNA"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Row-level split counts, task type, metric, prompt, and reported results were checked against the final creator paper.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction",
        "regression"
      ],
      "coverage_notes": [],
      "domains": [
        "transcriptomics",
        "multiomics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-sirna-closed-baselines",
        "bioinstruction-sirna-creator-systems",
        "bioinstruction-sirna-open-baselines"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-sirna-evidence-paper",
          "locator": {
            "note": "Task definition, split counts, input/output format, metric, and creator evaluation.",
            "type": "table",
            "value": "Table 2 (siRNA row); Appendix A.2-A.3; Table 8; Tables 4-7"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/parent_id",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/resources",
            "/implementations",
            "/versions/0/release_date",
            "/versions/0/as_of",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-sirna-evidence-repository",
          "locator": {
            "note": "Public grader implementation, partial artifact release, and absent repository license.",
            "type": "repository-path",
            "value": "evaluation/evaluate.py and evaluation/register_tasks.json at 600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "source_id": "bioinstruction-sirna-repository-resource",
          "source_type": "resource",
          "supports": [
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/implementations"
          ]
        }
      ],
      "field_status": [],
      "id": "bioinstruction-sirna",
      "implementations": [
        {
          "commit": "600acaa08c0302e8f5ce86de0fe041f21c13b53e",
          "framework": "Biology-Instructions official evaluator",
          "notes": "Task-specific siRNA output parsing and scoring.",
          "status": "official",
          "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e/evaluation"
        }
      ],
      "kind": "track",
      "latest_version": "emnlp-2025",
      "modalities": [
        "text",
        "dna-rna-sequence"
      ],
      "name": "Biology-Instructions siRNA Efficiency Prediction",
      "organizations": [
        "Shanghai Artificial Intelligence Laboratory",
        "University of Science and Technology of China",
        "University of Sydney",
        "University of Toronto",
        "Chinese University of Hong Kong",
        "Shanghai Jiao Tong University",
        "Fudan University",
        "Shanghai Innovation Institute"
      ],
      "parent_id": "bioinstruction",
      "release_date": "2024-12-26",
      "resources": [
        {
          "access_notes": "Creator paper Table 2 and siRNA task/result sections; PDF SHA256 3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a.",
          "id": "bioinstruction-sirna-paper-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-4.0",
          "pin": {
            "kind": "snapshot",
            "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf",
            "value": "sha256:3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a"
          },
          "type": "paper",
          "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf"
        },
        {
          "access_notes": "Official evaluator repository; no LICENSE file and no complete evaluation split at the pinned commit.",
          "id": "bioinstruction-sirna-repository-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e",
            "value": "600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "type": "repository",
          "url": "https://github.com/hhnqqq/Biology-Instructions"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "emnlp-2025",
        "entries": [
          {
            "confidence": "high",
            "count": 66987,
            "count_basis": "distinct examples across the published train, validation, and test splits",
            "count_ref": "/task_counts/total",
            "count_unit": "examples",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-sirna-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "siRNA efficiency prediction.",
            "reporting_status": "reported",
            "task_type_id": "sirna-efficacy-prediction"
          }
        ],
        "notes": "Single-purpose formal Biology-Instructions evaluation track.",
        "status": "complete"
      },
      "summary": "Regression of siRNA efficiency from paired sequence context, evaluated with the SAIS-inspired mixed score.",
      "task_counts": {
        "basis": "distinct examples across the published train, validation, and test splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "examples",
            "count": 53592,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-sirna-train",
            "label": "Training split",
            "notes": "Published training split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 6707,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-sirna-validation",
            "label": "Validation split",
            "notes": "Published validation split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 6688,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-sirna-test",
            "label": "Test split",
            "notes": "Held-out split used for creator-paper Tables 4-7.",
            "reporting_status": "reported"
          }
        ],
        "total": 66987
      },
      "task_formats": [
        "paired sequence regression"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task-level audit complete.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "bioinstruction-sirna-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "bioinstruction-sirna-emnlp-2025",
          "label": "emnlp-2025",
          "notes": "Final paper split: 53592 train / 6707 validation / 6688 test examples.",
          "release_date": "2025-11-04",
          "status": "current",
          "task_counts": {
            "basis": "distinct examples across the published train, validation, and test splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "examples",
                "count": 53592,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-sirna-train",
                "label": "Training split",
                "notes": "Published training split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 6707,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-sirna-validation",
                "label": "Validation split",
                "notes": "Published validation split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 6688,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-sirna-test",
                "label": "Test split",
                "notes": "Held-out split used for creator-paper Tables 4-7.",
                "reporting_status": "reported"
              }
            ],
            "total": 66987
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official post-processing evaluator is public, but the published test examples, labels, and model outputs are not included in the official repository.",
        "biosafety_notes": "BioBench Atlas stores task metadata and aggregate results only and does not mirror the upstream biological sequences.",
        "grader": "keyword/sentiment-assisted binary-label parser followed by accuracy.",
        "level": "partially-open",
        "license": null,
        "tasks": "The final paper publishes this task's source, definition, and split counts; the creator repository does not include the complete evaluation inputs and labels."
      },
      "aliases": [
        "Sol"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Row-level split counts, task type, metric, prompt, and reported results were checked against the final creator paper.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction",
        "classification"
      ],
      "coverage_notes": [],
      "domains": [
        "protein-sequence"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-solubility-closed-baselines",
        "bioinstruction-solubility-creator-systems",
        "bioinstruction-solubility-open-baselines"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-solubility-evidence-paper",
          "locator": {
            "note": "Task definition, split counts, input/output format, metric, and creator evaluation.",
            "type": "table",
            "value": "Table 2 (Sol row); Appendix A.2-A.3; Table 8; Tables 4-7"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/parent_id",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/resources",
            "/implementations",
            "/versions/0/release_date",
            "/versions/0/as_of",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-solubility-evidence-repository",
          "locator": {
            "note": "Public grader implementation, partial artifact release, and absent repository license.",
            "type": "repository-path",
            "value": "evaluation/evaluate.py and evaluation/register_tasks.json at 600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "source_id": "bioinstruction-solubility-repository-resource",
          "source_type": "resource",
          "supports": [
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/implementations"
          ]
        }
      ],
      "field_status": [],
      "id": "bioinstruction-solubility",
      "implementations": [
        {
          "commit": "600acaa08c0302e8f5ce86de0fe041f21c13b53e",
          "framework": "Biology-Instructions official evaluator",
          "notes": "Task-specific Sol output parsing and scoring.",
          "status": "official",
          "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e/evaluation"
        }
      ],
      "kind": "track",
      "latest_version": "emnlp-2025",
      "modalities": [
        "text",
        "protein-sequence"
      ],
      "name": "Biology-Instructions Protein Solubility Prediction",
      "organizations": [
        "Shanghai Artificial Intelligence Laboratory",
        "University of Science and Technology of China",
        "University of Sydney",
        "University of Toronto",
        "Chinese University of Hong Kong",
        "Shanghai Jiao Tong University",
        "Fudan University",
        "Shanghai Innovation Institute"
      ],
      "parent_id": "bioinstruction",
      "release_date": "2024-12-26",
      "resources": [
        {
          "access_notes": "Creator paper Table 2 and Sol task/result sections; PDF SHA256 3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a.",
          "id": "bioinstruction-solubility-paper-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-4.0",
          "pin": {
            "kind": "snapshot",
            "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf",
            "value": "sha256:3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a"
          },
          "type": "paper",
          "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf"
        },
        {
          "access_notes": "Official evaluator repository; no LICENSE file and no complete evaluation split at the pinned commit.",
          "id": "bioinstruction-solubility-repository-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e",
            "value": "600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "type": "repository",
          "url": "https://github.com/hhnqqq/Biology-Instructions"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "emnlp-2025",
        "entries": [
          {
            "confidence": "high",
            "count": 71421,
            "count_basis": "distinct examples across the published train, validation, and test splits",
            "count_ref": "/task_counts/total",
            "count_unit": "examples",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-solubility-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Protein solubility prediction.",
            "reporting_status": "reported",
            "task_type_id": "protein-solubility-prediction"
          }
        ],
        "notes": "Single-purpose formal Biology-Instructions evaluation track.",
        "status": "complete"
      },
      "summary": "Binary protein-solubility prediction from an amino-acid sequence, evaluated with accuracy.",
      "task_counts": {
        "basis": "distinct examples across the published train, validation, and test splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "examples",
            "count": 62478,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-solubility-train",
            "label": "Training split",
            "notes": "Published training split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 6942,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-solubility-validation",
            "label": "Validation split",
            "notes": "Published validation split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 2001,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-solubility-test",
            "label": "Test split",
            "notes": "Held-out split used for creator-paper Tables 4-7.",
            "reporting_status": "reported"
          }
        ],
        "total": 71421
      },
      "task_formats": [
        "protein-sequence binary classification"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task-level audit complete.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "bioinstruction-solubility-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "bioinstruction-solubility-emnlp-2025",
          "label": "emnlp-2025",
          "notes": "Final paper split: 62478 train / 6942 validation / 2001 test examples.",
          "release_date": "2025-11-04",
          "status": "current",
          "task_counts": {
            "basis": "distinct examples across the published train, validation, and test splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "examples",
                "count": 62478,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-solubility-train",
                "label": "Training split",
                "notes": "Published training split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 6942,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-solubility-validation",
                "label": "Validation split",
                "notes": "Published validation split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 2001,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-solubility-test",
                "label": "Test split",
                "notes": "Held-out split used for creator-paper Tables 4-7.",
                "reporting_status": "reported"
              }
            ],
            "total": 71421
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official post-processing evaluator is public, but the published test examples, labels, and model outputs are not included in the official repository.",
        "biosafety_notes": "BioBench Atlas stores task metadata and aggregate results only and does not mirror the upstream biological sequences.",
        "grader": "first-number parser followed by Spearman rank correlation.",
        "level": "partially-open",
        "license": null,
        "tasks": "The final paper publishes this task's source, definition, and split counts; the creator repository does not include the complete evaluation inputs and labels."
      },
      "aliases": [
        "Sta"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Row-level split counts, task type, metric, prompt, and reported results were checked against the final creator paper.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction",
        "regression"
      ],
      "coverage_notes": [],
      "domains": [
        "protein-sequence"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-stability-closed-baselines",
        "bioinstruction-stability-creator-systems",
        "bioinstruction-stability-open-baselines"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-stability-evidence-paper",
          "locator": {
            "note": "Task definition, split counts, input/output format, metric, and creator evaluation.",
            "type": "table",
            "value": "Table 2 (Sta row); Appendix A.2-A.3; Table 8; Tables 4-7"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/parent_id",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/resources",
            "/implementations",
            "/versions/0/release_date",
            "/versions/0/as_of",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-stability-evidence-repository",
          "locator": {
            "note": "Public grader implementation, partial artifact release, and absent repository license.",
            "type": "repository-path",
            "value": "evaluation/evaluate.py and evaluation/register_tasks.json at 600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "source_id": "bioinstruction-stability-repository-resource",
          "source_type": "resource",
          "supports": [
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/implementations"
          ]
        }
      ],
      "field_status": [],
      "id": "bioinstruction-stability",
      "implementations": [
        {
          "commit": "600acaa08c0302e8f5ce86de0fe041f21c13b53e",
          "framework": "Biology-Instructions official evaluator",
          "notes": "Task-specific Sta output parsing and scoring.",
          "status": "official",
          "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e/evaluation"
        }
      ],
      "kind": "track",
      "latest_version": "emnlp-2025",
      "modalities": [
        "text",
        "protein-sequence"
      ],
      "name": "Biology-Instructions Protein Stability Prediction",
      "organizations": [
        "Shanghai Artificial Intelligence Laboratory",
        "University of Science and Technology of China",
        "University of Sydney",
        "University of Toronto",
        "Chinese University of Hong Kong",
        "Shanghai Jiao Tong University",
        "Fudan University",
        "Shanghai Innovation Institute"
      ],
      "parent_id": "bioinstruction",
      "release_date": "2024-12-26",
      "resources": [
        {
          "access_notes": "Creator paper Table 2 and Sta task/result sections; PDF SHA256 3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a.",
          "id": "bioinstruction-stability-paper-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-4.0",
          "pin": {
            "kind": "snapshot",
            "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf",
            "value": "sha256:3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a"
          },
          "type": "paper",
          "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf"
        },
        {
          "access_notes": "Official evaluator repository; no LICENSE file and no complete evaluation split at the pinned commit.",
          "id": "bioinstruction-stability-repository-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e",
            "value": "600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "type": "repository",
          "url": "https://github.com/hhnqqq/Biology-Instructions"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "emnlp-2025",
        "entries": [
          {
            "confidence": "high",
            "count": 68977,
            "count_basis": "distinct examples across the published train, validation, and test splits",
            "count_ref": "/task_counts/total",
            "count_unit": "examples",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-stability-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Protein stability prediction.",
            "reporting_status": "reported",
            "task_type_id": "protein-stability-prediction"
          }
        ],
        "notes": "Single-purpose formal Biology-Instructions evaluation track.",
        "status": "complete"
      },
      "summary": "Regression of protein stability from an amino-acid sequence, evaluated with Spearman rank correlation.",
      "task_counts": {
        "basis": "distinct examples across the published train, validation, and test splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "examples",
            "count": 53614,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-stability-train",
            "label": "Training split",
            "notes": "Published training split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 2512,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-stability-validation",
            "label": "Validation split",
            "notes": "Published validation split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 12851,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-stability-test",
            "label": "Test split",
            "notes": "Held-out split used for creator-paper Tables 4-7.",
            "reporting_status": "reported"
          }
        ],
        "total": 68977
      },
      "task_formats": [
        "protein-sequence regression"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task-level audit complete.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "bioinstruction-stability-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "bioinstruction-stability-emnlp-2025",
          "label": "emnlp-2025",
          "notes": "Final paper split: 53614 train / 2512 validation / 12851 test examples.",
          "release_date": "2025-11-04",
          "status": "current",
          "task_counts": {
            "basis": "distinct examples across the published train, validation, and test splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "examples",
                "count": 53614,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-stability-train",
                "label": "Training split",
                "notes": "Published training split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 2512,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-stability-validation",
                "label": "Validation split",
                "notes": "Published validation split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 12851,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-stability-test",
                "label": "Test split",
                "notes": "Held-out split used for creator-paper Tables 4-7.",
                "reporting_status": "reported"
              }
            ],
            "total": 68977
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official post-processing evaluator is public, but the published test examples, labels, and model outputs are not included in the official repository.",
        "biosafety_notes": "BioBench Atlas stores task metadata and aggregate results only and does not mirror the upstream biological sequences.",
        "grader": "keyword/sentiment-assisted binary-label parser followed by Matthews correlation coefficient.",
        "level": "partially-open",
        "license": null,
        "tasks": "The final paper publishes this task's source, definition, and split counts; the creator repository does not include the complete evaluation inputs and labels."
      },
      "aliases": [
        "TB-H"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Row-level split counts, task type, metric, prompt, and reported results were checked against the final creator paper.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction",
        "classification"
      ],
      "coverage_notes": [],
      "domains": [
        "genomics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-tb-human-closed-baselines",
        "bioinstruction-tb-human-creator-systems",
        "bioinstruction-tb-human-open-baselines"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-tb-human-evidence-paper",
          "locator": {
            "note": "Task definition, split counts, input/output format, metric, and creator evaluation.",
            "type": "table",
            "value": "Table 2 (TB-H row); Appendix A.2-A.3; Table 8; Tables 4-7"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/parent_id",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/resources",
            "/implementations",
            "/versions/0/release_date",
            "/versions/0/as_of",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-tb-human-evidence-repository",
          "locator": {
            "note": "Public grader implementation, partial artifact release, and absent repository license.",
            "type": "repository-path",
            "value": "evaluation/evaluate.py and evaluation/register_tasks.json at 600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "source_id": "bioinstruction-tb-human-repository-resource",
          "source_type": "resource",
          "supports": [
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/implementations"
          ]
        }
      ],
      "field_status": [],
      "id": "bioinstruction-tb-human",
      "implementations": [
        {
          "commit": "600acaa08c0302e8f5ce86de0fe041f21c13b53e",
          "framework": "Biology-Instructions official evaluator",
          "notes": "Task-specific TB-H output parsing and scoring.",
          "status": "official",
          "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e/evaluation"
        }
      ],
      "kind": "track",
      "latest_version": "emnlp-2025",
      "modalities": [
        "text",
        "dna-rna-sequence"
      ],
      "name": "Biology-Instructions Human Transcription Binding Sites Detection",
      "organizations": [
        "Shanghai Artificial Intelligence Laboratory",
        "University of Science and Technology of China",
        "University of Sydney",
        "University of Toronto",
        "Chinese University of Hong Kong",
        "Shanghai Jiao Tong University",
        "Fudan University",
        "Shanghai Innovation Institute"
      ],
      "parent_id": "bioinstruction",
      "release_date": "2024-12-26",
      "resources": [
        {
          "access_notes": "Creator paper Table 2 and TB-H task/result sections; PDF SHA256 3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a.",
          "id": "bioinstruction-tb-human-paper-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-4.0",
          "pin": {
            "kind": "snapshot",
            "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf",
            "value": "sha256:3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a"
          },
          "type": "paper",
          "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf"
        },
        {
          "access_notes": "Official evaluator repository; no LICENSE file and no complete evaluation split at the pinned commit.",
          "id": "bioinstruction-tb-human-repository-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e",
            "value": "600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "type": "repository",
          "url": "https://github.com/hhnqqq/Biology-Instructions"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "emnlp-2025",
        "entries": [
          {
            "confidence": "high",
            "count": 138344,
            "count_basis": "distinct examples across the published train, validation, and test splits",
            "count_ref": "/task_counts/total",
            "count_unit": "examples",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-tb-human-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Human transcription-factor binding-site detection.",
            "reporting_status": "reported",
            "task_type_id": "transcription-factor-binding-site-prediction"
          }
        ],
        "notes": "Single-purpose formal Biology-Instructions evaluation track.",
        "status": "complete"
      },
      "summary": "Binary detection of transcription-factor binding sites in human DNA sequences, evaluated with Matthews correlation coefficient.",
      "task_counts": {
        "basis": "distinct examples across the published train, validation, and test splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "examples",
            "count": 128344,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-tb-human-train",
            "label": "Training split",
            "notes": "Published training split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 5000,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-tb-human-validation",
            "label": "Validation split",
            "notes": "Published validation split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 5000,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-tb-human-test",
            "label": "Test split",
            "notes": "Held-out split used for creator-paper Tables 4-7.",
            "reporting_status": "reported"
          }
        ],
        "total": 138344
      },
      "task_formats": [
        "human DNA-sequence binary classification"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task-level audit complete.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "bioinstruction-tb-human-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "bioinstruction-tb-human-emnlp-2025",
          "label": "emnlp-2025",
          "notes": "Final paper split: 128344 train / 5000 validation / 5000 test examples.",
          "release_date": "2025-11-04",
          "status": "current",
          "task_counts": {
            "basis": "distinct examples across the published train, validation, and test splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "examples",
                "count": 128344,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-tb-human-train",
                "label": "Training split",
                "notes": "Published training split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 5000,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-tb-human-validation",
                "label": "Validation split",
                "notes": "Published validation split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 5000,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-tb-human-test",
                "label": "Test split",
                "notes": "Held-out split used for creator-paper Tables 4-7.",
                "reporting_status": "reported"
              }
            ],
            "total": 138344
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official post-processing evaluator is public, but the published test examples, labels, and model outputs are not included in the official repository.",
        "biosafety_notes": "BioBench Atlas stores task metadata and aggregate results only and does not mirror the upstream biological sequences.",
        "grader": "keyword/sentiment-assisted binary-label parser followed by Matthews correlation coefficient.",
        "level": "partially-open",
        "license": null,
        "tasks": "The final paper publishes this task's source, definition, and split counts; the creator repository does not include the complete evaluation inputs and labels."
      },
      "aliases": [
        "TB-M"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Row-level split counts, task type, metric, prompt, and reported results were checked against the final creator paper.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction",
        "classification"
      ],
      "coverage_notes": [],
      "domains": [
        "genomics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-tb-mouse-closed-baselines",
        "bioinstruction-tb-mouse-creator-systems",
        "bioinstruction-tb-mouse-open-baselines"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-tb-mouse-evidence-paper",
          "locator": {
            "note": "Task definition, split counts, input/output format, metric, and creator evaluation.",
            "type": "table",
            "value": "Table 2 (TB-M row); Appendix A.2-A.3; Table 8; Tables 4-7"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/parent_id",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/resources",
            "/implementations",
            "/versions/0/release_date",
            "/versions/0/as_of",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-tb-mouse-evidence-repository",
          "locator": {
            "note": "Public grader implementation, partial artifact release, and absent repository license.",
            "type": "repository-path",
            "value": "evaluation/evaluate.py and evaluation/register_tasks.json at 600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "source_id": "bioinstruction-tb-mouse-repository-resource",
          "source_type": "resource",
          "supports": [
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/implementations"
          ]
        }
      ],
      "field_status": [],
      "id": "bioinstruction-tb-mouse",
      "implementations": [
        {
          "commit": "600acaa08c0302e8f5ce86de0fe041f21c13b53e",
          "framework": "Biology-Instructions official evaluator",
          "notes": "Task-specific TB-M output parsing and scoring.",
          "status": "official",
          "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e/evaluation"
        }
      ],
      "kind": "track",
      "latest_version": "emnlp-2025",
      "modalities": [
        "text",
        "dna-rna-sequence"
      ],
      "name": "Biology-Instructions Mouse Transcription Binding Sites Detection",
      "organizations": [
        "Shanghai Artificial Intelligence Laboratory",
        "University of Science and Technology of China",
        "University of Sydney",
        "University of Toronto",
        "Chinese University of Hong Kong",
        "Shanghai Jiao Tong University",
        "Fudan University",
        "Shanghai Innovation Institute"
      ],
      "parent_id": "bioinstruction",
      "release_date": "2024-12-26",
      "resources": [
        {
          "access_notes": "Creator paper Table 2 and TB-M task/result sections; PDF SHA256 3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a.",
          "id": "bioinstruction-tb-mouse-paper-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-4.0",
          "pin": {
            "kind": "snapshot",
            "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf",
            "value": "sha256:3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a"
          },
          "type": "paper",
          "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf"
        },
        {
          "access_notes": "Official evaluator repository; no LICENSE file and no complete evaluation split at the pinned commit.",
          "id": "bioinstruction-tb-mouse-repository-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e",
            "value": "600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "type": "repository",
          "url": "https://github.com/hhnqqq/Biology-Instructions"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "emnlp-2025",
        "entries": [
          {
            "confidence": "high",
            "count": 100028,
            "count_basis": "distinct examples across the published train, validation, and test splits",
            "count_ref": "/task_counts/total",
            "count_unit": "examples",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-tb-mouse-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Mouse transcription-factor binding-site detection.",
            "reporting_status": "reported",
            "task_type_id": "transcription-factor-binding-site-prediction"
          }
        ],
        "notes": "Single-purpose formal Biology-Instructions evaluation track.",
        "status": "complete"
      },
      "summary": "Binary detection of transcription-factor binding sites in mouse DNA sequences, evaluated with Matthews correlation coefficient.",
      "task_counts": {
        "basis": "distinct examples across the published train, validation, and test splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "examples",
            "count": 80018,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-tb-mouse-train",
            "label": "Training split",
            "notes": "Published training split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 10005,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-tb-mouse-validation",
            "label": "Validation split",
            "notes": "Published validation split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 10005,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-tb-mouse-test",
            "label": "Test split",
            "notes": "Held-out split used for creator-paper Tables 4-7.",
            "reporting_status": "reported"
          }
        ],
        "total": 100028
      },
      "task_formats": [
        "mouse DNA-sequence binary classification"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task-level audit complete.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "bioinstruction-tb-mouse-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "bioinstruction-tb-mouse-emnlp-2025",
          "label": "emnlp-2025",
          "notes": "Final paper split: 80018 train / 10005 validation / 10005 test examples.",
          "release_date": "2025-11-04",
          "status": "current",
          "task_counts": {
            "basis": "distinct examples across the published train, validation, and test splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "examples",
                "count": 80018,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-tb-mouse-train",
                "label": "Training split",
                "notes": "Published training split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 10005,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-tb-mouse-validation",
                "label": "Validation split",
                "notes": "Published validation split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 10005,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-tb-mouse-test",
                "label": "Test split",
                "notes": "Held-out split used for creator-paper Tables 4-7.",
                "reporting_status": "reported"
              }
            ],
            "total": 100028
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official post-processing evaluator is public, but the published test examples, labels, and model outputs are not included in the official repository.",
        "biosafety_notes": "BioBench Atlas stores task metadata and aggregate results only and does not mirror the upstream biological sequences.",
        "grader": "first-number parser followed by Spearman rank correlation.",
        "level": "partially-open",
        "license": null,
        "tasks": "The final paper publishes this task's source, definition, and split counts; the creator repository does not include the complete evaluation inputs and labels."
      },
      "aliases": [
        "Ther"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Row-level split counts, task type, metric, prompt, and reported results were checked against the final creator paper.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction",
        "regression"
      ],
      "coverage_notes": [],
      "domains": [
        "protein-sequence"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-thermostability-closed-baselines",
        "bioinstruction-thermostability-creator-systems",
        "bioinstruction-thermostability-open-baselines"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-thermostability-evidence-paper",
          "locator": {
            "note": "Task definition, split counts, input/output format, metric, and creator evaluation.",
            "type": "table",
            "value": "Table 2 (Ther row); Appendix A.2-A.3; Table 8; Tables 4-7"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/parent_id",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/resources",
            "/implementations",
            "/versions/0/release_date",
            "/versions/0/as_of",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-thermostability-evidence-repository",
          "locator": {
            "note": "Public grader implementation, partial artifact release, and absent repository license.",
            "type": "repository-path",
            "value": "evaluation/evaluate.py and evaluation/register_tasks.json at 600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "source_id": "bioinstruction-thermostability-repository-resource",
          "source_type": "resource",
          "supports": [
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/implementations"
          ]
        }
      ],
      "field_status": [],
      "id": "bioinstruction-thermostability",
      "implementations": [
        {
          "commit": "600acaa08c0302e8f5ce86de0fe041f21c13b53e",
          "framework": "Biology-Instructions official evaluator",
          "notes": "Task-specific Ther output parsing and scoring.",
          "status": "official",
          "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e/evaluation"
        }
      ],
      "kind": "track",
      "latest_version": "emnlp-2025",
      "modalities": [
        "text",
        "protein-sequence"
      ],
      "name": "Biology-Instructions Protein Thermostability Prediction",
      "organizations": [
        "Shanghai Artificial Intelligence Laboratory",
        "University of Science and Technology of China",
        "University of Sydney",
        "University of Toronto",
        "Chinese University of Hong Kong",
        "Shanghai Jiao Tong University",
        "Fudan University",
        "Shanghai Innovation Institute"
      ],
      "parent_id": "bioinstruction",
      "release_date": "2024-12-26",
      "resources": [
        {
          "access_notes": "Creator paper Table 2 and Ther task/result sections; PDF SHA256 3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a.",
          "id": "bioinstruction-thermostability-paper-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-4.0",
          "pin": {
            "kind": "snapshot",
            "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf",
            "value": "sha256:3bc92905728242b59bc660372f16c768433d69b64639e801e58e63f4e2b84d4a"
          },
          "type": "paper",
          "url": "https://aclanthology.org/2025.findings-emnlp.978.pdf"
        },
        {
          "access_notes": "Official evaluator repository; no LICENSE file and no complete evaluation split at the pinned commit.",
          "id": "bioinstruction-thermostability-repository-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://github.com/hhnqqq/Biology-Instructions/tree/600acaa08c0302e8f5ce86de0fe041f21c13b53e",
            "value": "600acaa08c0302e8f5ce86de0fe041f21c13b53e"
          },
          "type": "repository",
          "url": "https://github.com/hhnqqq/Biology-Instructions"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "emnlp-2025",
        "entries": [
          {
            "confidence": "high",
            "count": 7031,
            "count_basis": "distinct examples across the published train, validation, and test splits",
            "count_ref": "/task_counts/total",
            "count_unit": "examples",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bioinstruction-thermostability-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Protein thermostability prediction.",
            "reporting_status": "reported",
            "task_type_id": "protein-stability-prediction"
          }
        ],
        "notes": "Single-purpose formal Biology-Instructions evaluation track.",
        "status": "complete"
      },
      "summary": "Regression of protein thermostability from an amino-acid sequence, evaluated with Spearman rank correlation.",
      "task_counts": {
        "basis": "distinct examples across the published train, validation, and test splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "examples",
            "count": 5056,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-thermostability-train",
            "label": "Training split",
            "notes": "Published training split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 639,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-thermostability-validation",
            "label": "Validation split",
            "notes": "Published validation split.",
            "reporting_status": "reported"
          },
          {
            "basis": "examples",
            "count": 1336,
            "exclusive": true,
            "exhaustive": true,
            "id": "bioinstruction-thermostability-test",
            "label": "Test split",
            "notes": "Held-out split used for creator-paper Tables 4-7.",
            "reporting_status": "reported"
          }
        ],
        "total": 7031
      },
      "task_formats": [
        "protein-sequence regression"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task-level audit complete.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "bioinstruction-thermostability-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "bioinstruction-thermostability-emnlp-2025",
          "label": "emnlp-2025",
          "notes": "Final paper split: 5056 train / 639 validation / 1336 test examples.",
          "release_date": "2025-11-04",
          "status": "current",
          "task_counts": {
            "basis": "distinct examples across the published train, validation, and test splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "examples",
                "count": 5056,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-thermostability-train",
                "label": "Training split",
                "notes": "Published training split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 639,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-thermostability-validation",
                "label": "Validation split",
                "notes": "Published validation split.",
                "reporting_status": "reported"
              },
              {
                "basis": "examples",
                "count": 1336,
                "exclusive": true,
                "exhaustive": true,
                "id": "bioinstruction-thermostability-test",
                "label": "Test split",
                "notes": "Held-out split used for creator-paper Tables 4-7.",
                "reporting_status": "reported"
              }
            ],
            "total": 7031
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The preview includes 11 MB of anonymized data. The gated full release lists per-problem archives totaling about 155 GB and permits inference-time evaluation after accepting its conditions.",
        "biosafety_notes": "This registry mirrors no biological data, answers, or task files. The official materials describe anonymization and anti-lookup rules but do not publish a separate biosafety classification.",
        "grader": "Answer rubrics are included in the released problem tables and v11 specifies all-or-nothing scoring, but no executable official grading harness is published.",
        "level": "partially-open",
        "license": "CC BY 4.0 for problem statements, answer rubrics, and task formulation; data archives retain the original repositories' data-use policies and the full set has evaluation-only access conditions",
        "tasks": "A five-problem preview is open; the current 90-problem full set is gated behind evaluation-only terms that prohibit training, fine-tuning, reinforcement, or distillation use."
      },
      "aliases": [
        "Bio Mystery Bench"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "The current v11 data release is separated from the superseded v8 evaluation snapshot. Counts, access terms, printed result labels, and unreported protocol fields are not inferred across versions.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [
        "system-card-claude-opus-5-biomysterybench-1-use"
      ],
      "capabilities": [
        "data-analysis",
        "coding",
        "tool-use",
        "scientific-reasoning"
      ],
      "coverage_notes": [
        {
          "count": null,
          "coverage": "explicitly-in-scope",
          "notes": "The official report gives crystal-structure organism identification as an example, but no standalone structure-task count.",
          "reporting_status": "not_reported",
          "tag": "protein-structure"
        },
        {
          "count": null,
          "coverage": "explicitly-in-scope",
          "notes": "The official report explicitly lists scRNA-seq and gives a single-cell organ-identification example; no standalone count is published.",
          "reporting_status": "not_reported",
          "tag": "single-cell"
        },
        {
          "count": null,
          "coverage": "explicitly-in-scope",
          "notes": "The official report says several questions use proteomics, but does not publish a standalone count.",
          "reporting_status": "not_reported",
          "tag": "proteomics"
        },
        {
          "count": null,
          "coverage": "explicitly-in-scope",
          "notes": "The official report says several questions use metabolomics, but does not publish a standalone count.",
          "reporting_status": "not_reported",
          "tag": "metabolomics"
        }
      ],
      "domains": [
        "protein-structure",
        "genomics",
        "transcriptomics",
        "epigenomics",
        "single-cell",
        "proteomics",
        "metabolomics",
        "microbiome",
        "multiomics",
        "bioinformatics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "biomysterybench-official",
        "system-card-claude-opus-5"
      ],
      "evaluation_run_ids": [
        "biomysterybench-official-run",
        "biomysterybench-v8-human-difficult",
        "biomysterybench-v8-human-solvable"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "biomysterybench-evidence-identity",
          "locator": {
            "note": "Identifies Anthropic as creator and defines the method-agnostic, objective-answer agent benchmark.",
            "type": "section",
            "value": "Title, publication date, and “Benchmarking models on verifiable biological tasks with BioMysteryBench”"
          },
          "source_id": "biomysterybench-official",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/kind",
            "/summary",
            "/capabilities",
            "/task_formats"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "biomysterybench-evidence-v8-counts",
          "locator": {
            "note": "Reports the original 99 total, 76 human-solvable, 23 human-difficult, and four pre-release QC exclusions.",
            "type": "section",
            "value": "“Benchmarking models on verifiable biological tasks,” “Human-solvable,” and “Human-difficult”; v8 entry in the official dataset changelog"
          },
          "source_id": "biomysterybench-official",
          "source_type": "work",
          "supports": [
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "biomysterybench-evidence-v11",
          "locator": {
            "note": "Reports 99 to 90, the 73/17 partition, nine removals, 24 modified problems, and revised all-or-nothing rubric language.",
            "type": "repository-path",
            "value": "CHANGELOG.md, v11 (2026-07-06), at commit 51c9024021b8989a0cb06ae623b02f90d14c2da3"
          },
          "source_id": "biomysterybench-preview-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "biomysterybench-evidence-taxonomy",
          "locator": {
            "note": "Explicitly lists WGS, RNA-seq, scRNA-seq, methylation, ChIP-seq, metagenomics, Hi-C, proteomics, metabolomics, crystal structures, databases, and coding/tool use.",
            "type": "section",
            "value": "“Example questions” and the environment/property list"
          },
          "source_id": "biomysterybench-official",
          "source_type": "work",
          "supports": [
            "/domains",
            "/modalities",
            "/coverage_notes/0",
            "/coverage_notes/1",
            "/coverage_notes/2",
            "/coverage_notes/3",
            "/scientific_task_classification/entries/1"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "biomysterybench-evidence-access",
          "locator": {
            "note": "Documents the gated 90-problem release, evaluation-only restriction, CC BY 4.0 benchmark materials, source-policy data archives, and answer-rubric fields.",
            "type": "dataset-card",
            "value": "Access conditions, Contents, Rules, and License and terms of use at commit b5a889c4757214ec9a6ade876b734f920a7799db"
          },
          "source_id": "biomysterybench-full-resource",
          "source_type": "resource",
          "supports": [
            "/access/level",
            "/access/license",
            "/resources"
          ]
        }
      ],
      "field_status": [],
      "id": "biomysterybench",
      "implementations": [
        {
          "commit": null,
          "framework": "official containerized agent environment",
          "notes": "The benchmark inputs and rubrics are released, but no runner, environment image, or executable grader was identified in the official repositories.",
          "status": "not-available",
          "url": null
        }
      ],
      "kind": "agentic-eval",
      "latest_version": "v11",
      "modalities": [
        "text",
        "table",
        "dna-rna-sequence",
        "structure-3d",
        "raw-omics",
        "database",
        "code"
      ],
      "name": "BioMysteryBench",
      "organizations": [
        "Anthropic"
      ],
      "parent_id": null,
      "release_date": "2026-04-29",
      "resources": [
        {
          "access_notes": "Official creator report describing the original 99-problem v8 evaluation, environment, human baseline, results, and reliability analysis.",
          "id": "biomysterybench-report-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "snapshot",
            "url": "https://www.anthropic.com/research/Evaluating-Claude-For-Bioinformatics-With-BioMysteryBench",
            "value": "sha256:cb8b5d2e4a006dacfebbc53fd204a39b75e0e637343527c50518414db371eb6f"
          },
          "type": "website",
          "url": "https://www.anthropic.com/research/Evaluating-Claude-For-Bioinformatics-With-BioMysteryBench"
        },
        {
          "access_notes": "Open five-problem v11 preview with problem table, answer rubrics, allowed domains, human-solvability flags, and anonymized data.",
          "id": "biomysterybench-preview-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY 4.0",
          "pin": {
            "kind": "commit",
            "url": "https://huggingface.co/datasets/Anthropic/BioMysteryBench-preview/tree/51c9024021b8989a0cb06ae623b02f90d14c2da3",
            "value": "51c9024021b8989a0cb06ae623b02f90d14c2da3"
          },
          "type": "dataset",
          "url": "https://huggingface.co/datasets/Anthropic/BioMysteryBench-preview"
        },
        {
          "access_notes": "Gated 90-problem v11 full set; access requires accepting evaluation-only and no-training conditions.",
          "id": "biomysterybench-full-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY 4.0 for benchmark materials; source-repository policies for data archives",
          "pin": {
            "kind": "commit",
            "url": "https://huggingface.co/datasets/Anthropic/BioMysteryBench-full/tree/b5a889c4757214ec9a6ade876b734f920a7799db",
            "value": "b5a889c4757214ec9a6ade876b734f920a7799db"
          },
          "type": "dataset",
          "url": "https://huggingface.co/datasets/Anthropic/BioMysteryBench-full"
        },
        {
          "access_notes": "Official Figure 1 image with printed human-solvable scores and bootstrap error bars.",
          "id": "biomysterybench-figure-one-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "snapshot",
            "url": "https://www-cdn.anthropic.com/images/4zrzovbb/website/291077c3785708a54dbb4421db3751b1d1c4ba84-1920x1080.png",
            "value": "sha256:aa3f7b8a7f84c5fad6f9a1adb9e884687800ca93017083de72ae06da637f7a1b"
          },
          "type": "documentation",
          "url": "https://www-cdn.anthropic.com/images/4zrzovbb/website/291077c3785708a54dbb4421db3751b1d1c4ba84-1920x1080.png"
        },
        {
          "access_notes": "Official Figure 2 image with printed human-difficult scores and bootstrap error bars.",
          "id": "biomysterybench-figure-two-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "snapshot",
            "url": "https://www-cdn.anthropic.com/images/4zrzovbb/website/0aed26a846ccd67d813a2c14f78069216671c8e3-1920x1080.png",
            "value": "sha256:fa2066f08f54b4ca55500a022b8661fc3cd5af495770462edd1ed2358e7067e7"
          },
          "type": "documentation",
          "url": "https://www-cdn.anthropic.com/images/4zrzovbb/website/0aed26a846ccd67d813a2c14f78069216671c8e3-1920x1080.png"
        },
        {
          "access_notes": "Official Figure 3 image showing the per-problem 0-of-5 through 5-of-5 solve-count distributions.",
          "id": "biomysterybench-figure-three-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "snapshot",
            "url": "https://www-cdn.anthropic.com/images/4zrzovbb/website/3840cd87380589c9f00c31d5b9334639ebfb5303-1920x1080.png",
            "value": "sha256:5da53558196ce4bc1493bd59e80acac72c9360a79aaf0aeb1b001cf11ad28ab9"
          },
          "type": "documentation",
          "url": "https://www-cdn.anthropic.com/images/4zrzovbb/website/3840cd87380589c9f00c31d5b9334639ebfb5303-1920x1080.png"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "v11",
        "entries": [
          {
            "confidence": "high",
            "count": 90,
            "count_basis": "v11 mystery-bioinformatics problems after the June 2026 answer-key audit",
            "count_ref": "/task_counts/total",
            "count_unit": "problems",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "biomysterybench-evidence-v11"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "Each mystery is scored on its final answer rather than a prescribed analysis path.",
            "reporting_status": "reported",
            "task_type_id": "end-to-end-computational-analysis"
          },
          {
            "confidence": "high",
            "count": null,
            "count_basis": "v11 mystery-bioinformatics problems using omics and cellular data.",
            "count_ref": null,
            "count_unit": "problems",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "biomysterybench-evidence-taxonomy"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "Single-cell, proteomics, metabolomics, and other modalities are explicit, without a leaf-task count.",
            "reporting_status": "not_reported",
            "task_type_id": "omics-cellular-analysis"
          }
        ],
        "notes": "Official modality examples support a broad end-to-end analysis mapping but not exhaustive task-topic counts.",
        "status": "partial"
      },
      "summary": "An agentic bioinformatics benchmark of objective, expert-authored mysteries over anonymized real-world biological data, scored on final answers rather than prescribed analysis paths.",
      "task_counts": {
        "basis": "v11 mystery-bioinformatics problems after the June 2026 answer-key audit",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "problems solved by at least one human benchmarker",
            "count": 73,
            "exclusive": true,
            "exhaustive": true,
            "id": "human-solvable",
            "label": "Human-solvable (v11)",
            "notes": "The v11 changelog reports 73 human-solvable problems.",
            "reporting_status": "reported"
          },
          {
            "basis": "problems not solved by the human panel",
            "count": 17,
            "exclusive": true,
            "exhaustive": true,
            "id": "human-difficult",
            "label": "Human-hard (v11)",
            "notes": "The v11 changelog calls this split human-hard; the April evaluation report used human-difficult for its 23-problem predecessor.",
            "reporting_status": "reported"
          }
        ],
        "total": 90
      },
      "task_formats": [
        "open-ended bioinformatics investigation",
        "containerized agent episode"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Field-level audit completed against the official Anthropic report and commit-pinned Hugging Face v11 releases.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-04-28",
          "evidence_ids": [
            "biomysterybench-evidence-v8-counts"
          ],
          "formal_tracks": [],
          "id": "biomysterybench-v8",
          "label": "v8",
          "notes": "Initial public release used by the April creator evaluation. The v11 changelog identifies it as v8 and notes that 16 problems had accession-ID leaks scrubbed.",
          "release_date": "2026-04-28",
          "status": "superseded",
          "task_counts": {
            "basis": "initial public-release problems after pre-release QC exclusions",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "problems solved by at least one human benchmarker",
                "count": 76,
                "exclusive": true,
                "exhaustive": true,
                "id": "human-solvable",
                "label": "Human-solvable (v8)",
                "notes": "Up to five domain experts attempted each question; the split includes questions solved by at least one.",
                "reporting_status": "reported"
              },
              {
                "basis": "problems unsolved by the expert panel after removing four failed-QC questions",
                "count": 23,
                "exclusive": true,
                "exhaustive": true,
                "id": "human-difficult",
                "label": "Human-difficult (v8)",
                "notes": "Four malformed, broken, or inherently unsolvable candidates were removed before this 23-problem split was reported.",
                "reporting_status": "reported"
              }
            ],
            "total": 99
          }
        },
        {
          "as_of": "2026-07-06",
          "evidence_ids": [
            "biomysterybench-evidence-v11"
          ],
          "formal_tracks": [],
          "id": "biomysterybench-v11",
          "label": "v11",
          "notes": "Nine problems were removed and 24 were modified after an answer-key audit with expert reruns, model reruns, and underlying-data checks; the data files for modified problems did not change.",
          "release_date": "2026-07-06",
          "status": "current",
          "task_counts": {
            "basis": "v11 mystery-bioinformatics problems after the June 2026 answer-key audit",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "problems solved by at least one human benchmarker",
                "count": 73,
                "exclusive": true,
                "exhaustive": true,
                "id": "human-solvable",
                "label": "Human-solvable (v11)",
                "notes": "The v11 changelog reports 73 human-solvable problems.",
                "reporting_status": "reported"
              },
              {
                "basis": "problems not solved by the human panel",
                "count": 17,
                "exclusive": true,
                "exhaustive": true,
                "id": "human-difficult",
                "label": "Human-hard (v11)",
                "notes": "The v11 changelog calls this split human-hard; the April evaluation report used human-difficult for its 23-problem predecessor.",
                "reporting_status": "reported"
              }
            ],
            "total": 90
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The public subset provides example task prompts and descriptive metadata.",
        "biosafety_notes": "The entire evaluation set is restricted in accordance with biosecurity research practice.",
        "grader": "Not established by the independently verified metadata claims.",
        "level": "partially-open",
        "license": null,
        "tasks": "Example evaluation task prompts are public; the entire evaluation set is restricted."
      },
      "aliases": [],
      "audit": {
        "audited_date": "2026-07-30",
        "notes": "Automated double-pass extraction plus deterministic official-resource pin; the benchmark license remains unverified and is published as null.",
        "status": "audited-with-caveats",
        "unresolved_fields": 1
      },
      "benchmark_use_ids": [
        "biosecbench-8d53fd8-claude-code-use",
        "biosecbench-8d53fd8-openai-codex-use",
        "biosecbench-8d53fd8-pi-use",
        "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-1-use",
        "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-2-use",
        "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-3-use",
        "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-4-use"
      ],
      "capabilities": [
        "data-analysis",
        "tool-use",
        "scientific-reasoning"
      ],
      "coverage_notes": [],
      "domains": [
        "genomics",
        "bioinformatics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
        "biosecbench-surveillance-repository-result-snapshot"
      ],
      "evaluation_run_ids": [
        "biosecbench-8d53fd8-claude-code",
        "biosecbench-8d53fd8-openai-codex",
        "biosecbench-8d53fd8-pi"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-metadata-1-evidence",
          "locator": {
            "document_page": 7,
            "note": null,
            "printed_page": "7",
            "source_fragment_sha256": "c947b9b0491115c064306906bbba7432d3ec36b057f8dd1c027260cdd2ee6172",
            "type": "section",
            "value": "Data availability"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/access/artifacts"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-metadata-2-evidence",
          "locator": {
            "document_page": 7,
            "note": null,
            "printed_page": "7",
            "source_fragment_sha256": "8d5ac92cd97b3e3993bdb2f83f5d68b38e4193f898d160d7867db824978f1f73",
            "type": "section",
            "value": "Data availability"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/access/biosafety_notes"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-metadata-3-evidence",
          "locator": {
            "document_page": 7,
            "note": null,
            "printed_page": "7",
            "source_fragment_sha256": "2d95b087029e38f56519d50d5197e229d3d720050f5fccdf3d3efe617d53c722",
            "type": "section",
            "value": "Data availability"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/access/level"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-metadata-4-evidence",
          "locator": {
            "document_page": 7,
            "note": null,
            "printed_page": "7",
            "source_fragment_sha256": "9cd4d1ce62593287bedd6351e1698615f63a717b9bc010a8c1fe3feca01667e9",
            "type": "section",
            "value": "Data availability"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/access/tasks"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-metadata-5-evidence",
          "locator": {
            "document_page": 2,
            "note": null,
            "printed_page": "2",
            "source_fragment_sha256": "8835f16f60e06e05995a6650e6d60bcfb21169401505974dd811c4d3fa2a592b",
            "type": "section",
            "value": "Introduction"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/capabilities"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-metadata-6-evidence",
          "locator": {
            "document_page": 7,
            "note": null,
            "printed_page": "7",
            "source_fragment_sha256": "2175cef09a0e89671f622f1cbf12dc41f881e1180963caf18f835dabb19597b4",
            "type": "section",
            "value": "Methods — Benchmark composition and data; Agent runs and execution"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/domains"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-metadata-7-evidence",
          "locator": {
            "document_page": 2,
            "note": null,
            "printed_page": "2",
            "source_fragment_sha256": "86d44cdc55b40b47d05305f247776ce518d25e2bc3cae72be542cc4ba2c79443",
            "type": "section",
            "value": "Introduction"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/kind"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-metadata-8-evidence",
          "locator": {
            "document_page": 7,
            "note": null,
            "printed_page": "7",
            "source_fragment_sha256": "be85dd21fbf2f268b4b997b75cad9f3c7b60fef92261adfe511e9c4151ad98c8",
            "type": "section",
            "value": "Methods — Benchmark composition and data; Agent runs and execution"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/modalities"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-metadata-9-evidence",
          "locator": {
            "document_page": 1,
            "note": null,
            "printed_page": "1",
            "source_fragment_sha256": "d8062a766e37ec1f470ba23e7973e72f44eff09d69f99acc298bd07bc84a5312",
            "type": "section",
            "value": "Abstract"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/name"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-metadata-10-evidence",
          "locator": {
            "document_page": 1,
            "note": null,
            "printed_page": "1",
            "source_fragment_sha256": "b3f87859f204d156ce37bf698c31ab426d939c240a7f9fa4762ee50f2a81dec6",
            "type": "page",
            "value": "Author-affiliation mapping"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/organizations"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-metadata-11-evidence",
          "locator": {
            "document_page": 1,
            "note": null,
            "printed_page": "1",
            "source_fragment_sha256": "611c54567b0f65f16bd9fdc25e80295903a4ee3dec4d99a40784f05b71bc5e74",
            "type": "page",
            "value": "arXiv dateline"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/release_date"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-metadata-12-evidence",
          "locator": {
            "document_page": 2,
            "note": null,
            "printed_page": "2",
            "source_fragment_sha256": "7fa81a7e9913a2a1faa7b321edf3474c9bd40d12e7629050d3d71733eaa3b3b3",
            "type": "section",
            "value": "Introduction and Benchmark construction"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/summary"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-metadata-13-evidence",
          "locator": {
            "document_page": 7,
            "note": null,
            "printed_page": "7",
            "source_fragment_sha256": "4dc267c1494cdaf0dbac3b360d2f399bfc5da4d52e6ffdc5358441f6ca0967f3",
            "type": "section",
            "value": "Methods — Task format and deterministic grading"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/task_formats"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-count-evidence",
          "locator": {
            "document_page": 1,
            "note": null,
            "printed_page": "1",
            "source_fragment_sha256": "eefd011459f343e68cbac3b78fcbdcad94bce0f2c65a1ffc2261ce90118ef237",
            "type": "section",
            "value": "Abstract"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/task_counts/total",
            "/task_counts/basis",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-subset-1-evidence",
          "locator": {
            "document_page": 3,
            "note": null,
            "printed_page": "3",
            "source_fragment_sha256": "60836cfbcaa2ac2b5eac4c17b429c650cd84efc98ba192bcdc538d9faff70479",
            "type": "figure",
            "value": "Figure 1A"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/task_counts/subsets",
            "/versions/0/task_counts/subsets",
            "/task_counts/subsets/0",
            "/versions/0/task_counts/subsets/0"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-subset-2-evidence",
          "locator": {
            "document_page": 3,
            "note": null,
            "printed_page": "3",
            "source_fragment_sha256": "827b7f283ec2f8827b885aa0d92020cdc4ea9b488a48c15329bf8d2ee17901f4",
            "type": "figure",
            "value": "Figure 1A"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/task_counts/subsets/1",
            "/versions/0/task_counts/subsets/1"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-subset-3-evidence",
          "locator": {
            "document_page": 3,
            "note": null,
            "printed_page": "3",
            "source_fragment_sha256": "4afdfb8b3dcf7c0f6c58fd8477d8d586fb3f66523094e881ef74137dc2103e9e",
            "type": "figure",
            "value": "Figure 1A"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/task_counts/subsets/2",
            "/versions/0/task_counts/subsets/2"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-subset-4-evidence",
          "locator": {
            "document_page": 3,
            "note": null,
            "printed_page": "3",
            "source_fragment_sha256": "ee51f80f42f513f6bc159db8e2d9b70ebb4fb0e36071d101e212863c7784f32f",
            "type": "figure",
            "value": "Figure 1A"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/task_counts/subsets/3",
            "/versions/0/task_counts/subsets/3"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-subset-5-evidence",
          "locator": {
            "document_page": 3,
            "note": null,
            "printed_page": "3",
            "source_fragment_sha256": "f09556c299711390ce8e2765d322c0e7d4b9bba4071d668d7971dba8b9e048bb",
            "type": "figure",
            "value": "Figure 1A"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/task_counts/subsets/4",
            "/versions/0/task_counts/subsets/4"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-subset-6-evidence",
          "locator": {
            "document_page": 3,
            "note": null,
            "printed_page": "3",
            "source_fragment_sha256": "ba8d8027957cd81c1b1c45d77da11cf957d548dae41447a228f733b87e52c163",
            "type": "figure",
            "value": "Figure 1A"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/task_counts/subsets/5",
            "/versions/0/task_counts/subsets/5"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-subset-7-evidence",
          "locator": {
            "document_page": 3,
            "note": null,
            "printed_page": "3",
            "source_fragment_sha256": "436755afc4d989b87fe6a8fda4e5fc924c51a37b8f74f6f729eb4fbba17390f2",
            "type": "figure",
            "value": "Figure 1A"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/task_counts/subsets/6",
            "/versions/0/task_counts/subsets/6"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-subset-8-evidence",
          "locator": {
            "document_page": 3,
            "note": null,
            "printed_page": "3",
            "source_fragment_sha256": "b13e4f3434eab7abe1c2841793562b0aabf5d6c808e05ebb86973dcd1fcab037",
            "type": "figure",
            "value": "Figure 1B"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/task_counts/subsets/7",
            "/versions/0/task_counts/subsets/7"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-subset-9-evidence",
          "locator": {
            "document_page": 3,
            "note": null,
            "printed_page": "3",
            "source_fragment_sha256": "2f266a64a2b67de4a0439f635cb417c8410fe90296e305556100665c9f01bdcc",
            "type": "figure",
            "value": "Figure 1B"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/task_counts/subsets/8",
            "/versions/0/task_counts/subsets/8"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-subset-10-evidence",
          "locator": {
            "document_page": 3,
            "note": null,
            "printed_page": "3",
            "source_fragment_sha256": "70a902f44c4e07c6b897eadf1c27972b2a7e2dc483bf65579f2be31d13de8def",
            "type": "figure",
            "value": "Figure 1B"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/task_counts/subsets/9",
            "/versions/0/task_counts/subsets/9"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-subset-11-evidence",
          "locator": {
            "document_page": 3,
            "note": null,
            "printed_page": "3",
            "source_fragment_sha256": "040fc853412a43620e02a51166cb29984754d1682f6950e4399bad55994388be",
            "type": "figure",
            "value": "Figure 1B"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/task_counts/subsets/10",
            "/versions/0/task_counts/subsets/10"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-subset-12-evidence",
          "locator": {
            "document_page": 3,
            "note": null,
            "printed_page": "3",
            "source_fragment_sha256": "8320b02f52d36922aa5fdc9996b4dd09432aeef951a3996951d655ca711959e4",
            "type": "figure",
            "value": "Figure 1B"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/task_counts/subsets/11",
            "/versions/0/task_counts/subsets/11"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-subset-13-evidence",
          "locator": {
            "document_page": 3,
            "note": null,
            "printed_page": "3",
            "source_fragment_sha256": "1c56c78949b9251aa16d18ca28841d6047c86974e77635245b0ff3af7ccb3f9a",
            "type": "figure",
            "value": "Figure 1B"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/task_counts/subsets/12",
            "/versions/0/task_counts/subsets/12"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-subset-14-evidence",
          "locator": {
            "document_page": 3,
            "note": null,
            "printed_page": "3",
            "source_fragment_sha256": "97c7e8c729915d3ef0cb5643b2ddb7246133e3f7c58e7253514ee3eda02651d0",
            "type": "figure",
            "value": "Figure 1C"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/task_counts/subsets/13",
            "/versions/0/task_counts/subsets/13"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-subset-15-evidence",
          "locator": {
            "document_page": 3,
            "note": null,
            "printed_page": "3",
            "source_fragment_sha256": "826617f2ea16addb52179bf5031a4f5c532f8733718539c50f7ef4c06a5fcd6e",
            "type": "figure",
            "value": "Figure 1C"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/task_counts/subsets/14",
            "/versions/0/task_counts/subsets/14"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-subset-16-evidence",
          "locator": {
            "document_page": 3,
            "note": null,
            "printed_page": "3",
            "source_fragment_sha256": "70be78fd5a68d5daa47eef601c446860e154160dcae4b744c5c65391ec8666b0",
            "type": "figure",
            "value": "Figure 1C"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/task_counts/subsets/15",
            "/versions/0/task_counts/subsets/15"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-subset-17-evidence",
          "locator": {
            "document_page": 3,
            "note": null,
            "printed_page": "3",
            "source_fragment_sha256": "4667a5475a548a0670691c363608560ae85a110d7b1a4e90645863813e1f5be2",
            "type": "figure",
            "value": "Figure 1D"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/task_counts/subsets/16",
            "/versions/0/task_counts/subsets/16"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-subset-18-evidence",
          "locator": {
            "document_page": 3,
            "note": null,
            "printed_page": "3",
            "source_fragment_sha256": "910118da0c5bbde931cc144359ee7b63aa1ab5967c1c94dc381660cb5f06b860",
            "type": "figure",
            "value": "Figure 1D"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/task_counts/subsets/17",
            "/versions/0/task_counts/subsets/17"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-subset-19-evidence",
          "locator": {
            "document_page": 3,
            "note": null,
            "printed_page": "3",
            "source_fragment_sha256": "163ad2f0100627f2295a9c88ac8c90b5b26c0331d8ba81fb76335e6e00bcfa5a",
            "type": "figure",
            "value": "Figure 1D"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/task_counts/subsets/18",
            "/versions/0/task_counts/subsets/18"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-subset-20-evidence",
          "locator": {
            "document_page": 3,
            "note": null,
            "printed_page": "3",
            "source_fragment_sha256": "632c750c1dbc880c43a145d3840967bb6c146c91dc9ce7c0fc9e2cd2e4ca72a9",
            "type": "figure",
            "value": "Figure 1E"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/task_counts/subsets/19",
            "/versions/0/task_counts/subsets/19"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-subset-21-evidence",
          "locator": {
            "document_page": 3,
            "note": null,
            "printed_page": "3",
            "source_fragment_sha256": "3e3a2a4eb6f71c599e2bfb71245d58402572ae2d139777538afad818cc4cc2a5",
            "type": "figure",
            "value": "Figure 1E"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/task_counts/subsets/20",
            "/versions/0/task_counts/subsets/20"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-resource-evidence",
          "locator": {
            "document_page": 7,
            "note": null,
            "printed_page": "7",
            "source_fragment_sha256": "ee8499693e901477a5198e6a3af425c4d0b06f7d99978b10cfc058b922196f14",
            "type": "section",
            "value": "Data availability"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/resources",
            "/implementations",
            "/access/license"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-creator-evidence",
          "locator": {
            "document_page": 1,
            "note": null,
            "printed_page": "1",
            "source_fragment_sha256": "611c54567b0f65f16bd9fdc25e80295903a4ee3dec4d99a40784f05b71bc5e74",
            "type": "page",
            "value": "arXiv dateline"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/resources/0"
          ]
        },
        {
          "accessed_date": "2026-07-30",
          "id": "biosecbench-surveillance-automated-version-evidence",
          "locator": {
            "document_page": 1,
            "note": null,
            "printed_page": "1",
            "source_fragment_sha256": "611c54567b0f65f16bd9fdc25e80295903a4ee3dec4d99a40784f05b71bc5e74",
            "type": "page",
            "value": "arXiv dateline"
          },
          "source_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
          "source_type": "work",
          "supports": [
            "/latest_version",
            "/versions/0"
          ]
        }
      ],
      "field_status": [
        {
          "confidence": "high",
          "evidence_ids": [
            "biosecbench-surveillance-automated-resource-evidence"
          ],
          "path": "/access/license",
          "reason": "The double-pass review verified the official resource identity but did not establish a redistributable benchmark license; the value remains null pending source-level license verification.",
          "status": "provisional"
        }
      ],
      "id": "biosecbench-surveillance",
      "implementations": [
        {
          "commit": "8d53fd8517cc74202eb18b618e8b39b4ffaf0c87",
          "framework": "official repository",
          "notes": "Commit resolved deterministically during paper intake.",
          "status": "official",
          "url": "https://github.com/latchbio/biosecbench-surveillance"
        }
      ],
      "kind": "agentic-eval",
      "latest_version": "initial-release",
      "modalities": [
        "text",
        "dna-rna-sequence",
        "raw-omics",
        "database"
      ],
      "name": "BioSecBench-Surveillance",
      "organizations": [
        "LatchBio",
        "Aclid"
      ],
      "parent_id": null,
      "release_date": "2026-07-21",
      "resources": [
        {
          "access_notes": "Versioned creator source; full text is not mirrored.",
          "id": "biosecbench-surveillance-creator-paper-resource",
          "last_checked": "2026-07-30",
          "license": null,
          "pin": null,
          "type": "paper",
          "url": "https://arxiv.org/abs/2607.19262"
        },
        {
          "access_notes": "Official resource pinned immutably during intake.",
          "id": "biosecbench-surveillance-official-repository-resource",
          "last_checked": "2026-07-30",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://github.com/latchbio/biosecbench-surveillance/commit/8d53fd8517cc74202eb18b618e8b39b4ffaf0c87",
            "value": "8d53fd8517cc74202eb18b618e8b39b4ffaf0c87"
          },
          "type": "repository",
          "url": "https://github.com/latchbio/biosecbench-surveillance"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "initial-release",
        "entries": [],
        "notes": "No Scientific Task claim passed independent high-confidence verification; task mapping remains pending a targeted official-source audit.",
        "status": "partial"
      },
      "summary": "Agentic evaluation of pathogen genomic-surveillance workflow selection and analysis from raw or near-raw sequencing data.",
      "task_counts": {
        "basis": "The source explicitly defines the benchmark as 100 evaluations.",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "Printed Figure 1A bar label.",
            "count": 23,
            "exclusive": true,
            "exhaustive": true,
            "id": "task-category-variant-detection",
            "label": "Task category — Variant detection evaluations",
            "notes": null,
            "partition_group": "task-category",
            "reporting_status": "reported"
          },
          {
            "basis": "Printed Figure 1A bar label.",
            "count": 16,
            "exclusive": true,
            "exhaustive": true,
            "id": "task-category-taxonomic-classification",
            "label": "Task category — Taxonomic classification evaluations",
            "notes": null,
            "partition_group": "task-category",
            "reporting_status": "reported"
          },
          {
            "basis": "Printed Figure 1A bar label.",
            "count": 15,
            "exclusive": true,
            "exhaustive": true,
            "id": "task-category-amr-characterization",
            "label": "Task category — AMR characterization evaluations",
            "notes": null,
            "partition_group": "task-category",
            "reporting_status": "reported"
          },
          {
            "basis": "Printed Figure 1A bar label.",
            "count": 12,
            "exclusive": true,
            "exhaustive": true,
            "id": "task-category-toxin-and-virulence-characterization",
            "label": "Task category — Toxin and virulence characterization evaluations",
            "notes": null,
            "partition_group": "task-category",
            "reporting_status": "reported"
          },
          {
            "basis": "Printed Figure 1A bar label.",
            "count": 12,
            "exclusive": true,
            "exhaustive": true,
            "id": "task-category-genetic-engineering-characterization",
            "label": "Task category — Genetic-engineering characterization evaluations",
            "notes": null,
            "partition_group": "task-category",
            "reporting_status": "reported"
          },
          {
            "basis": "Printed Figure 1A bar label.",
            "count": 12,
            "exclusive": true,
            "exhaustive": true,
            "id": "task-category-source-tracking",
            "label": "Task category — Source tracking evaluations",
            "notes": null,
            "partition_group": "task-category",
            "reporting_status": "reported"
          },
          {
            "basis": "Printed Figure 1A bar label.",
            "count": 10,
            "exclusive": true,
            "exhaustive": true,
            "id": "task-category-anomaly-detection",
            "label": "Task category — Anomaly detection evaluations",
            "notes": null,
            "partition_group": "task-category",
            "reporting_status": "reported"
          },
          {
            "basis": "Printed Figure 1B bar label.",
            "count": 34,
            "exclusive": true,
            "exhaustive": true,
            "id": "sample-type-isolate",
            "label": "Sample type — Isolate evaluations",
            "notes": null,
            "partition_group": "sample-type",
            "reporting_status": "reported"
          },
          {
            "basis": "Printed Figure 1B bar label.",
            "count": 31,
            "exclusive": true,
            "exhaustive": true,
            "id": "sample-type-wastewater",
            "label": "Sample type — Wastewater evaluations",
            "notes": null,
            "partition_group": "sample-type",
            "reporting_status": "reported"
          },
          {
            "basis": "Printed Figure 1B bar label.",
            "count": 23,
            "exclusive": true,
            "exhaustive": true,
            "id": "sample-type-clinical",
            "label": "Sample type — Clinical evaluations",
            "notes": null,
            "partition_group": "sample-type",
            "reporting_status": "reported"
          },
          {
            "basis": "Printed Figure 1B bar label.",
            "count": 5,
            "exclusive": true,
            "exhaustive": true,
            "id": "sample-type-agricultural",
            "label": "Sample type — Agricultural evaluations",
            "notes": null,
            "partition_group": "sample-type",
            "reporting_status": "reported"
          },
          {
            "basis": "Printed Figure 1B bar label.",
            "count": 5,
            "exclusive": true,
            "exhaustive": true,
            "id": "sample-type-air",
            "label": "Sample type — Air evaluations",
            "notes": null,
            "partition_group": "sample-type",
            "reporting_status": "reported"
          },
          {
            "basis": "Printed Figure 1B bar label.",
            "count": 2,
            "exclusive": true,
            "exhaustive": true,
            "id": "sample-type-water",
            "label": "Sample type — Water evaluations",
            "notes": null,
            "partition_group": "sample-type",
            "reporting_status": "reported"
          },
          {
            "basis": "Printed Figure 1C bar label.",
            "count": 80,
            "exclusive": true,
            "exhaustive": true,
            "id": "sequencing-technology-short-read",
            "label": "Sequencing technology — Short-read evaluations",
            "notes": null,
            "partition_group": "sequencing-technology",
            "reporting_status": "reported"
          },
          {
            "basis": "Printed Figure 1C bar label.",
            "count": 15,
            "exclusive": true,
            "exhaustive": true,
            "id": "sequencing-technology-long-read",
            "label": "Sequencing technology — Long-read evaluations",
            "notes": null,
            "partition_group": "sequencing-technology",
            "reporting_status": "reported"
          },
          {
            "basis": "Printed Figure 1C bar label.",
            "count": 5,
            "exclusive": true,
            "exhaustive": true,
            "id": "sequencing-technology-hybrid",
            "label": "Sequencing technology — Hybrid evaluations",
            "notes": null,
            "partition_group": "sequencing-technology",
            "reporting_status": "reported"
          },
          {
            "basis": "Printed Figure 1D bar label.",
            "count": 64,
            "exclusive": true,
            "exhaustive": true,
            "id": "nucleic-acid-target-dna",
            "label": "Nucleic-acid target — DNA evaluations",
            "notes": null,
            "partition_group": "nucleic-acid-target",
            "reporting_status": "reported"
          },
          {
            "basis": "Printed Figure 1D bar label.",
            "count": 33,
            "exclusive": true,
            "exhaustive": true,
            "id": "nucleic-acid-target-rna",
            "label": "Nucleic-acid target — RNA evaluations",
            "notes": null,
            "partition_group": "nucleic-acid-target",
            "reporting_status": "reported"
          },
          {
            "basis": "Printed Figure 1D bar label.",
            "count": 3,
            "exclusive": true,
            "exhaustive": true,
            "id": "nucleic-acid-target-total-na",
            "label": "Nucleic-acid target — Total NA evaluations",
            "notes": null,
            "partition_group": "nucleic-acid-target",
            "reporting_status": "reported"
          },
          {
            "basis": "Printed Figure 1E bar label.",
            "count": 85,
            "exclusive": true,
            "exhaustive": true,
            "id": "assay-type-shotgun",
            "label": "Assay type — Shotgun evaluations",
            "notes": null,
            "partition_group": "assay-type",
            "reporting_status": "reported"
          },
          {
            "basis": "Figure 1 defines Targeted as targeted/amplicon/capture.",
            "count": 15,
            "exclusive": true,
            "exhaustive": true,
            "id": "assay-type-targeted-amplicon-capture",
            "label": "Assay type — Targeted/amplicon/capture evaluations",
            "notes": null,
            "partition_group": "assay-type",
            "reporting_status": "reported"
          }
        ],
        "total": 100
      },
      "task_formats": [
        "agentic sequencing-data analysis with structured JSON answers"
      ],
      "verification": {
        "last_verified": "2026-07-30",
        "notes": "New family admitted after creator source and official resource verification; the unresolved benchmark license is visibly flagged.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": null,
          "evidence_ids": [
            "biosecbench-surveillance-automated-count-evidence",
            "biosecbench-surveillance-automated-subset-1-evidence",
            "biosecbench-surveillance-automated-subset-2-evidence",
            "biosecbench-surveillance-automated-subset-3-evidence",
            "biosecbench-surveillance-automated-subset-4-evidence",
            "biosecbench-surveillance-automated-subset-5-evidence",
            "biosecbench-surveillance-automated-subset-6-evidence",
            "biosecbench-surveillance-automated-subset-7-evidence",
            "biosecbench-surveillance-automated-subset-8-evidence",
            "biosecbench-surveillance-automated-subset-9-evidence",
            "biosecbench-surveillance-automated-subset-10-evidence",
            "biosecbench-surveillance-automated-subset-11-evidence",
            "biosecbench-surveillance-automated-subset-12-evidence",
            "biosecbench-surveillance-automated-subset-13-evidence",
            "biosecbench-surveillance-automated-subset-14-evidence",
            "biosecbench-surveillance-automated-subset-15-evidence",
            "biosecbench-surveillance-automated-subset-16-evidence",
            "biosecbench-surveillance-automated-subset-17-evidence",
            "biosecbench-surveillance-automated-subset-18-evidence",
            "biosecbench-surveillance-automated-subset-19-evidence",
            "biosecbench-surveillance-automated-subset-20-evidence",
            "biosecbench-surveillance-automated-subset-21-evidence",
            "biosecbench-surveillance-automated-version-evidence"
          ],
          "formal_tracks": [],
          "id": "biosecbench-surveillance-initial-release-version",
          "label": "initial-release",
          "notes": "Atlas snapshot label for the creator-paper release; the source does not report a formal benchmark version.",
          "release_date": "2026-07-21",
          "status": "current",
          "task_counts": {
            "basis": "The source explicitly defines the benchmark as 100 evaluations.",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "Printed Figure 1A bar label.",
                "count": 23,
                "exclusive": true,
                "exhaustive": true,
                "id": "task-category-variant-detection",
                "label": "Task category — Variant detection evaluations",
                "notes": null,
                "partition_group": "task-category",
                "reporting_status": "reported"
              },
              {
                "basis": "Printed Figure 1A bar label.",
                "count": 16,
                "exclusive": true,
                "exhaustive": true,
                "id": "task-category-taxonomic-classification",
                "label": "Task category — Taxonomic classification evaluations",
                "notes": null,
                "partition_group": "task-category",
                "reporting_status": "reported"
              },
              {
                "basis": "Printed Figure 1A bar label.",
                "count": 15,
                "exclusive": true,
                "exhaustive": true,
                "id": "task-category-amr-characterization",
                "label": "Task category — AMR characterization evaluations",
                "notes": null,
                "partition_group": "task-category",
                "reporting_status": "reported"
              },
              {
                "basis": "Printed Figure 1A bar label.",
                "count": 12,
                "exclusive": true,
                "exhaustive": true,
                "id": "task-category-toxin-and-virulence-characterization",
                "label": "Task category — Toxin and virulence characterization evaluations",
                "notes": null,
                "partition_group": "task-category",
                "reporting_status": "reported"
              },
              {
                "basis": "Printed Figure 1A bar label.",
                "count": 12,
                "exclusive": true,
                "exhaustive": true,
                "id": "task-category-genetic-engineering-characterization",
                "label": "Task category — Genetic-engineering characterization evaluations",
                "notes": null,
                "partition_group": "task-category",
                "reporting_status": "reported"
              },
              {
                "basis": "Printed Figure 1A bar label.",
                "count": 12,
                "exclusive": true,
                "exhaustive": true,
                "id": "task-category-source-tracking",
                "label": "Task category — Source tracking evaluations",
                "notes": null,
                "partition_group": "task-category",
                "reporting_status": "reported"
              },
              {
                "basis": "Printed Figure 1A bar label.",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "task-category-anomaly-detection",
                "label": "Task category — Anomaly detection evaluations",
                "notes": null,
                "partition_group": "task-category",
                "reporting_status": "reported"
              },
              {
                "basis": "Printed Figure 1B bar label.",
                "count": 34,
                "exclusive": true,
                "exhaustive": true,
                "id": "sample-type-isolate",
                "label": "Sample type — Isolate evaluations",
                "notes": null,
                "partition_group": "sample-type",
                "reporting_status": "reported"
              },
              {
                "basis": "Printed Figure 1B bar label.",
                "count": 31,
                "exclusive": true,
                "exhaustive": true,
                "id": "sample-type-wastewater",
                "label": "Sample type — Wastewater evaluations",
                "notes": null,
                "partition_group": "sample-type",
                "reporting_status": "reported"
              },
              {
                "basis": "Printed Figure 1B bar label.",
                "count": 23,
                "exclusive": true,
                "exhaustive": true,
                "id": "sample-type-clinical",
                "label": "Sample type — Clinical evaluations",
                "notes": null,
                "partition_group": "sample-type",
                "reporting_status": "reported"
              },
              {
                "basis": "Printed Figure 1B bar label.",
                "count": 5,
                "exclusive": true,
                "exhaustive": true,
                "id": "sample-type-agricultural",
                "label": "Sample type — Agricultural evaluations",
                "notes": null,
                "partition_group": "sample-type",
                "reporting_status": "reported"
              },
              {
                "basis": "Printed Figure 1B bar label.",
                "count": 5,
                "exclusive": true,
                "exhaustive": true,
                "id": "sample-type-air",
                "label": "Sample type — Air evaluations",
                "notes": null,
                "partition_group": "sample-type",
                "reporting_status": "reported"
              },
              {
                "basis": "Printed Figure 1B bar label.",
                "count": 2,
                "exclusive": true,
                "exhaustive": true,
                "id": "sample-type-water",
                "label": "Sample type — Water evaluations",
                "notes": null,
                "partition_group": "sample-type",
                "reporting_status": "reported"
              },
              {
                "basis": "Printed Figure 1C bar label.",
                "count": 80,
                "exclusive": true,
                "exhaustive": true,
                "id": "sequencing-technology-short-read",
                "label": "Sequencing technology — Short-read evaluations",
                "notes": null,
                "partition_group": "sequencing-technology",
                "reporting_status": "reported"
              },
              {
                "basis": "Printed Figure 1C bar label.",
                "count": 15,
                "exclusive": true,
                "exhaustive": true,
                "id": "sequencing-technology-long-read",
                "label": "Sequencing technology — Long-read evaluations",
                "notes": null,
                "partition_group": "sequencing-technology",
                "reporting_status": "reported"
              },
              {
                "basis": "Printed Figure 1C bar label.",
                "count": 5,
                "exclusive": true,
                "exhaustive": true,
                "id": "sequencing-technology-hybrid",
                "label": "Sequencing technology — Hybrid evaluations",
                "notes": null,
                "partition_group": "sequencing-technology",
                "reporting_status": "reported"
              },
              {
                "basis": "Printed Figure 1D bar label.",
                "count": 64,
                "exclusive": true,
                "exhaustive": true,
                "id": "nucleic-acid-target-dna",
                "label": "Nucleic-acid target — DNA evaluations",
                "notes": null,
                "partition_group": "nucleic-acid-target",
                "reporting_status": "reported"
              },
              {
                "basis": "Printed Figure 1D bar label.",
                "count": 33,
                "exclusive": true,
                "exhaustive": true,
                "id": "nucleic-acid-target-rna",
                "label": "Nucleic-acid target — RNA evaluations",
                "notes": null,
                "partition_group": "nucleic-acid-target",
                "reporting_status": "reported"
              },
              {
                "basis": "Printed Figure 1D bar label.",
                "count": 3,
                "exclusive": true,
                "exhaustive": true,
                "id": "nucleic-acid-target-total-na",
                "label": "Nucleic-acid target — Total NA evaluations",
                "notes": null,
                "partition_group": "nucleic-acid-target",
                "reporting_status": "reported"
              },
              {
                "basis": "Printed Figure 1E bar label.",
                "count": 85,
                "exclusive": true,
                "exhaustive": true,
                "id": "assay-type-shotgun",
                "label": "Assay type — Shotgun evaluations",
                "notes": null,
                "partition_group": "assay-type",
                "reporting_status": "reported"
              },
              {
                "basis": "Figure 1 defines Targeted as targeted/amplicon/capture.",
                "count": 15,
                "exclusive": true,
                "exhaustive": true,
                "id": "assay-type-targeted-amplicon-capture",
                "label": "Assay type — Targeted/amplicon/capture evaluations",
                "notes": null,
                "partition_group": "assay-type",
                "reporting_status": "reported"
              }
            ],
            "total": 100
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "Public capsule archives contain the notebook-derived analysis environments and associated data; users must authenticate to Hugging Face to download them.",
        "biosafety_notes": "The benchmark covers general computational-biology analyses and publishes a no-training canary; this registry mirrors neither task data nor capsule archives, and the creators do not publish a separate biosafety classification.",
        "grader": "The Apache-2.0 repository publishes the string, range, LLM, multiple-choice, and open-ended grading code.",
        "level": "fully-open",
        "license": "Apache-2.0 for the official code and current dataset; the preprint's own reuse license is not reported",
        "tasks": "The official Hugging Face release publishes all 205 v1.5 questions, targets, distractors, verifier modes, and canaries."
      },
      "aliases": [
        "Bioinformatics Benchmark"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Version, question, verifier, taxonomy, access, protocol, and result claims were audited against creator sources. The upstream 60-notebook claim cannot be reconciled to the 59 referenced capsule UUIDs and 64 stored archives.",
        "status": "audited-with-caveats",
        "unresolved_fields": 2
      },
      "benchmark_use_ids": [
        "anthropic-life-sciences-bixbench"
      ],
      "capabilities": [
        "data-analysis",
        "coding",
        "tool-use",
        "scientific-reasoning"
      ],
      "coverage_notes": [
        {
          "count": 74,
          "coverage": "observed",
          "notes": "v1.5 rows carrying the multi-label Genomics category; category counts overlap.",
          "reporting_status": "reported",
          "tag": "genomics"
        },
        {
          "count": 69,
          "coverage": "observed",
          "notes": "v1.5 rows carrying the multi-label Transcriptomics category; category counts overlap.",
          "reporting_status": "reported",
          "tag": "transcriptomics"
        },
        {
          "count": 12,
          "coverage": "observed",
          "notes": "v1.5 rows carrying the multi-label Epigenomics category; category counts overlap.",
          "reporting_status": "reported",
          "tag": "epigenomics"
        },
        {
          "count": 2,
          "coverage": "observed",
          "notes": "v1.5 rows carrying the multi-label Single-Cell category; category counts overlap.",
          "reporting_status": "reported",
          "tag": "single-cell"
        },
        {
          "count": 4,
          "coverage": "observed",
          "notes": "v1.5 rows carrying the multi-label Proteomics category; category counts overlap.",
          "reporting_status": "reported",
          "tag": "proteomics"
        },
        {
          "count": 2,
          "coverage": "observed",
          "notes": "v1.5 rows carrying the multi-label Integrative Omics category; category counts overlap.",
          "reporting_status": "reported",
          "tag": "multiomics"
        },
        {
          "count": null,
          "coverage": "unknown",
          "notes": "No standalone protein-protein-binding count is published or encoded as a v1.5 category.",
          "reporting_status": "not_reported",
          "tag": "protein-protein-binding"
        },
        {
          "count": null,
          "coverage": "unknown",
          "notes": "No standalone protein-ligand-binding count is published or encoded as a v1.5 category.",
          "reporting_status": "not_reported",
          "tag": "protein-ligand-binding"
        }
      ],
      "domains": [
        "bioinformatics",
        "genomics",
        "transcriptomics",
        "epigenomics",
        "single-cell",
        "proteomics",
        "multiomics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "anthropic-life-sciences",
        "bixbench-paper",
        "bixbench-v1-5-release"
      ],
      "evaluation_run_ids": [
        "bixbench-creator-paper",
        "bixbench-paper-mcq-no-images",
        "bixbench-paper-mcq-no-refusal",
        "bixbench-paper-mcq-refusal",
        "bixbench-v1-5-agentic-mcq-no-refusal-images",
        "bixbench-v1-5-agentic-mcq-refusal-images",
        "bixbench-v1-5-agentic-mcq-refusal-no-images",
        "bixbench-v1-5-agentic-open-images",
        "bixbench-v1-5-zero-shot-mcq-no-refusal",
        "bixbench-v1-5-zero-shot-mcq-refusal",
        "bixbench-v1-5-zero-shot-open"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bixbench-evidence-identity",
          "locator": {
            "note": "Names the benchmark, creators, initial publication, agentic data-analysis objective, and open/MCQ modes.",
            "type": "page",
            "value": "PDF pp. 1–2, title, abstract, and contributions"
          },
          "source_id": "bixbench-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/kind",
            "/summary",
            "/capabilities",
            "/task_formats"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bixbench-evidence-paper-counts",
          "locator": {
            "note": "Reports 53 analytical scenarios/capsules and 296 questions in the original release.",
            "type": "page",
            "value": "PDF pp. 2, 4, and 6, Contributions, §3.2.1, and §4.1"
          },
          "source_id": "bixbench-paper",
          "source_type": "work",
          "supports": [
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bixbench-evidence-v1-5-counts",
          "locator": {
            "note": "205 unique rows; verifier modes are 83 llm_verifier, 61 str_verifier, and 61 range_verifier.",
            "type": "repository-path",
            "value": "BixBench.jsonl at commit 2f660cbcab36b87f5d49b997f71e67be94131889"
          },
          "source_id": "bixbench-v1-5-dataset-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bixbench-evidence-capsule-count-conflict",
          "locator": {
            "note": "README claims 60 source notebooks; JSONL references 59 capsule UUIDs; the tagged tree stores 64 CapsuleFolder ZIP files.",
            "type": "repository-path",
            "value": "README.md, BixBench.jsonl, and v1.5 tree at commit 2f660cbcab36b87f5d49b997f71e67be94131889"
          },
          "source_id": "bixbench-v1-5-dataset-resource",
          "source_type": "resource",
          "supports": [
            "/task_counts/subsets",
            "/task_counts/subsets/1/count",
            "/versions/1/task_counts/subsets",
            "/versions/1/task_counts/subsets/1/count"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bixbench-evidence-taxonomy",
          "locator": {
            "note": "Multi-label category counts include Genomics 74, Transcriptomics 69, Epigenomics 12, Proteomics 4, Single-Cell 2, and Integrative Omics 2; rows also reference tabular, sequence, imaging, raw-data, and notebook analyses.",
            "type": "repository-path",
            "value": "BixBench.jsonl category, question, and artifact fields"
          },
          "source_id": "bixbench-v1-5-dataset-resource",
          "source_type": "resource",
          "supports": [
            "/domains",
            "/modalities",
            "/coverage_notes",
            "/scientific_task_classification/entries/1"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bixbench-evidence-access-license",
          "locator": {
            "note": "Apache-2.0 license and public Docker harness, prompts, graders, and download/run instructions.",
            "type": "repository-path",
            "value": "README.md, LICENSE, bixbench/graders.py, and scripts at commit 49311180bdacb324c596f2e07596c126f2004008"
          },
          "source_id": "bixbench-repository-resource",
          "source_type": "resource",
          "supports": [
            "/access/level",
            "/access/license",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/resources",
            "/implementations"
          ]
        }
      ],
      "field_status": [
        {
          "confidence": "high",
          "evidence_ids": [
            "bixbench-evidence-capsule-count-conflict"
          ],
          "path": "/task_counts/subsets/1/count",
          "reason": "The README states 60 published notebooks/capsules, while v1.5 rows reference 59 unique capsule UUIDs and the release tree contains 64 capsule ZIP files; the relation among these units is undocumented.",
          "status": "conflicted"
        },
        {
          "confidence": "high",
          "evidence_ids": [
            "bixbench-evidence-capsule-count-conflict"
          ],
          "path": "/versions/1/task_counts/subsets/1/count",
          "reason": "The README states 60 published notebooks/capsules, while v1.5 rows reference 59 unique capsule UUIDs and the release tree contains 64 capsule ZIP files; the relation among these units is undocumented.",
          "status": "conflicted"
        }
      ],
      "id": "bixbench",
      "implementations": [
        {
          "commit": "49311180bdacb324c596f2e07596c126f2004008",
          "framework": "BixBench Aviary agent and zero-shot harness",
          "notes": "Current Docker runner, prompt templates, graders, and result processing.",
          "status": "official",
          "url": "https://github.com/Future-House/BixBench/tree/49311180bdacb324c596f2e07596c126f2004008"
        },
        {
          "commit": "6c28217959d5d7dd6f48c59894534fced7c6c040",
          "framework": "BixBench paper-era Aviary harness",
          "notes": "Snapshot aligned to the original 296-question publication.",
          "status": "official",
          "url": "https://github.com/Future-House/BixBench/tree/6c28217959d5d7dd6f48c59894534fced7c6c040"
        }
      ],
      "kind": "agentic-eval",
      "latest_version": "v1.5",
      "modalities": [
        "text",
        "table",
        "figure",
        "dna-rna-sequence",
        "raw-omics",
        "image",
        "code"
      ],
      "name": "BixBench",
      "organizations": [
        "FutureHouse",
        "ScienceMachine"
      ],
      "parent_id": null,
      "release_date": "2025-02-28",
      "resources": [
        {
          "access_notes": "Creator-hosted preprint PDF describing the original 296-question, 53-capsule v1.0 evaluation.",
          "id": "bixbench-paper-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "snapshot",
            "url": "https://storage.googleapis.com/bixbench-results/BixBench.pdf",
            "value": "sha256:784d36d0b16e24479f3fe86614b089b57cc5e27be075723bc4736652ec92e637"
          },
          "type": "paper",
          "url": "https://storage.googleapis.com/bixbench-results/BixBench.pdf"
        },
        {
          "access_notes": "Official code, current harness, graders, release results, and reproduction scripts.",
          "id": "bixbench-repository-resource",
          "last_checked": "2026-07-21",
          "license": "Apache-2.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Future-House/BixBench/tree/49311180bdacb324c596f2e07596c126f2004008",
            "value": "49311180bdacb324c596f2e07596c126f2004008"
          },
          "type": "repository",
          "url": "https://github.com/Future-House/BixBench"
        },
        {
          "access_notes": "Paper-era harness snapshot with the 296-question postprocessing configuration and 25-step ReAct settings.",
          "id": "bixbench-v1-paper-code-resource",
          "last_checked": "2026-07-21",
          "license": "Apache-2.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Future-House/BixBench/tree/6c28217959d5d7dd6f48c59894534fced7c6c040",
            "value": "6c28217959d5d7dd6f48c59894534fced7c6c040"
          },
          "type": "repository",
          "url": "https://github.com/Future-House/BixBench"
        },
        {
          "access_notes": "Official v1.5 tag; BixBench.jsonl has SHA256 0d1204dcdae7193a9132ced5a3502008f6b3b163debc1b65b2aa2d86cb132dc9.",
          "id": "bixbench-v1-5-dataset-resource",
          "last_checked": "2026-07-21",
          "license": "Apache-2.0",
          "pin": {
            "kind": "commit",
            "url": "https://huggingface.co/datasets/futurehouse/BixBench/tree/2f660cbcab36b87f5d49b997f71e67be94131889",
            "value": "2f660cbcab36b87f5d49b997f71e67be94131889"
          },
          "type": "dataset",
          "url": "https://huggingface.co/datasets/futurehouse/BixBench"
        },
        {
          "access_notes": "Official v1.5 result release, including exact zero-shot baseline JSON and plots for agentic conditions.",
          "id": "bixbench-v1-5-results-resource",
          "last_checked": "2026-07-21",
          "license": "Apache-2.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Future-House/BixBench/tree/28909d842bc492ecd99bab303279afb29e3cb353/bixbench-v1.5_results",
            "value": "28909d842bc492ecd99bab303279afb29e3cb353"
          },
          "type": "documentation",
          "url": "https://github.com/Future-House/BixBench/tree/28909d842bc492ecd99bab303279afb29e3cb353/bixbench-v1.5_results"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "v1.5",
        "entries": [
          {
            "confidence": "high",
            "count": 205,
            "count_basis": "one question per row in the official v1.5 BixBench.jsonl",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "bixbench-evidence-v1-5-counts"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "Questions are grounded in containerized published analysis capsules.",
            "reporting_status": "reported",
            "task_type_id": "end-to-end-computational-analysis"
          },
          {
            "confidence": "high",
            "count": null,
            "count_basis": "v1.5 questions carrying official genomics, transcriptomics, epigenomics, single-cell, proteomics, or integrative-omics category labels.",
            "count_ref": null,
            "count_unit": "questions",
            "coverage": "observed",
            "evidence_ids": [
              "bixbench-evidence-taxonomy"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "Official multi-label domain categories overlap, so no additive omics question total is asserted.",
            "reporting_status": "not_reported",
            "task_type_id": "omics-cellular-analysis"
          }
        ],
        "notes": "Official categories are multi-label domains rather than an exhaustive scientific-task taxonomy.",
        "status": "partial"
      },
      "summary": "A containerized benchmark of long-horizon bioinformatics analysis over real published notebooks and associated data, with open-answer and multiple-choice evaluation modes.",
      "task_counts": {
        "basis": "one question per row in the official v1.5 BixBench.jsonl",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "unique capsule_uuid values referenced by the 205 rows",
            "count": 59,
            "exclusive": false,
            "exhaustive": false,
            "id": "bixbench-v1-5-referenced-capsules",
            "label": "Referenced capsules in v1.5 rows",
            "notes": "A source-artifact count, not a partition of questions.",
            "reporting_status": "reported"
          },
          {
            "basis": "real-world published Jupyter notebooks and related capsules stated in the official README",
            "count": 60,
            "exclusive": false,
            "exhaustive": false,
            "id": "bixbench-v1-5-readme-notebooks",
            "label": "Source notebooks stated in README",
            "notes": "Conflicts with 59 referenced capsule UUIDs and 64 capsule archives; upstream does not document how the three units correspond.",
            "reporting_status": "reported"
          },
          {
            "basis": "CapsuleFolder-*.zip paths in the official v1.5 dataset tag",
            "count": 64,
            "exclusive": false,
            "exhaustive": false,
            "id": "bixbench-v1-5-release-archives",
            "label": "Capsule archives in v1.5 release",
            "notes": "A stored-artifact count, not a partition of questions.",
            "reporting_status": "reported"
          },
          {
            "basis": "rows whose eval_mode is llm_verifier",
            "count": 83,
            "exclusive": true,
            "exhaustive": true,
            "id": "bixbench-v1-5-llm-verifier",
            "label": "LLM-verifier questions",
            "notes": "Mutually exclusive with the two deterministic verifier modes.",
            "reporting_status": "reported"
          },
          {
            "basis": "rows whose eval_mode is str_verifier",
            "count": 61,
            "exclusive": true,
            "exhaustive": true,
            "id": "bixbench-v1-5-string-verifier",
            "label": "String-verifier questions",
            "notes": "Mutually exclusive with the other verifier modes.",
            "reporting_status": "reported"
          },
          {
            "basis": "rows whose eval_mode is range_verifier",
            "count": 61,
            "exclusive": true,
            "exhaustive": true,
            "id": "bixbench-v1-5-range-verifier",
            "label": "Range-verifier questions",
            "notes": "Mutually exclusive with the other verifier modes.",
            "reporting_status": "reported"
          }
        ],
        "total": 205
      },
      "task_formats": [
        "open-ended bioinformatics analysis",
        "multiple choice"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Audited against the creator PDF, Apache-2.0 repositories, commit-pinned v1.5 dataset, graders, and official result JSON; conflicting source-artifact units remain explicitly flagged.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2025-03-03",
          "evidence_ids": [
            "bixbench-evidence-paper-counts"
          ],
          "formal_tracks": [],
          "id": "bixbench-v1-0",
          "label": "v1.0",
          "notes": "Original paper snapshot used for the creator's ten-trajectory agent evaluation; superseded after substantial question review and flattening in v1.5.",
          "release_date": "2025-02-28",
          "status": "superseded",
          "task_counts": {
            "basis": "open-answer questions reported in the original creator preprint",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "expert-curated notebook-and-data scenarios in the paper release",
                "count": 53,
                "exclusive": false,
                "exhaustive": false,
                "id": "bixbench-v1-capsules",
                "label": "Analytical scenarios or capsules",
                "notes": "A scenario count, not a partition of the 296 questions.",
                "reporting_status": "reported"
              }
            ],
            "total": 296
          }
        },
        {
          "as_of": "2025-09-29",
          "evidence_ids": [
            "bixbench-evidence-v1-5-counts",
            "bixbench-evidence-capsule-count-conflict"
          ],
          "formal_tracks": [],
          "id": "bixbench-v1-5",
          "label": "v1.5",
          "notes": "Current flattened release after substantial re-review and revision of the original questions; the v1.0 tag remains available upstream.",
          "release_date": "2025-09-23",
          "status": "current",
          "task_counts": {
            "basis": "one question per row in the official v1.5 BixBench.jsonl",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "unique capsule_uuid values referenced by the 205 rows",
                "count": 59,
                "exclusive": false,
                "exhaustive": false,
                "id": "bixbench-v1-5-referenced-capsules",
                "label": "Referenced capsules in v1.5 rows",
                "notes": "A source-artifact count, not a partition of questions.",
                "reporting_status": "reported"
              },
              {
                "basis": "real-world published Jupyter notebooks and related capsules stated in the official README",
                "count": 60,
                "exclusive": false,
                "exhaustive": false,
                "id": "bixbench-v1-5-readme-notebooks",
                "label": "Source notebooks stated in README",
                "notes": "Conflicts with 59 referenced capsule UUIDs and 64 capsule archives; upstream does not document how the three units correspond.",
                "reporting_status": "reported"
              },
              {
                "basis": "CapsuleFolder-*.zip paths in the official v1.5 dataset tag",
                "count": 64,
                "exclusive": false,
                "exhaustive": false,
                "id": "bixbench-v1-5-release-archives",
                "label": "Capsule archives in v1.5 release",
                "notes": "A stored-artifact count, not a partition of questions.",
                "reporting_status": "reported"
              },
              {
                "basis": "rows whose eval_mode is llm_verifier",
                "count": 83,
                "exclusive": true,
                "exhaustive": true,
                "id": "bixbench-v1-5-llm-verifier",
                "label": "LLM-verifier questions",
                "notes": "Mutually exclusive with the two deterministic verifier modes.",
                "reporting_status": "reported"
              },
              {
                "basis": "rows whose eval_mode is str_verifier",
                "count": 61,
                "exclusive": true,
                "exhaustive": true,
                "id": "bixbench-v1-5-string-verifier",
                "label": "String-verifier questions",
                "notes": "Mutually exclusive with the other verifier modes.",
                "reporting_status": "reported"
              },
              {
                "basis": "rows whose eval_mode is range_verifier",
                "count": 61,
                "exclusive": true,
                "exhaustive": true,
                "id": "bixbench-v1-5-range-verifier",
                "label": "Range-verifier questions",
                "notes": "Mutually exclusive with the other verifier modes.",
                "reporting_status": "reported"
              }
            ],
            "total": 205
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The repository publishes expert annotations and grader inputs; code is Apache-2.0 and the benchmark data directory is ODC-By-1.0.",
        "biosafety_notes": "Four source questions concern ecology, primate behavior, or evolutionary biology, but BLADE contains no protein, omics, pathogen, intervention, or wet-lab task and the creators publish no separate biosafety restriction.",
        "grader": "Public code performs executable-code checks, value and graph matching for transformations, and GPT-4o-assisted semantic conversion and matching for conceptual variables and statistical models.",
        "level": "fully-open",
        "license": "Apache-2.0 for code; ODC-By-1.0 for the benchmark data and annotations; CC BY 4.0 for the ACL paper",
        "tasks": "The official repository publishes the twelve research questions, data tables, 188 MCQs, and associated dataset descriptions."
      },
      "aliases": [
        "Benchmarking Language Model Agents for Data-Driven Science"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Identity, counts, formal task types, taxonomy, licenses, runnable implementation, grader, protocols, and printed generation results were checked against creator sources. Distinct count units are retained instead of being summed into a synthetic suite total.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "knowledge",
        "classification",
        "data-analysis",
        "coding",
        "tool-use",
        "scientific-reasoning"
      ],
      "coverage_notes": [
        {
          "count": 4,
          "coverage": "observed",
          "notes": "Registry mapping of four explicitly biological or ecological source questions in paper Table 3: AMTL, Crofoot, Panda_nuts, and Fish. BLADE itself is cross-domain.",
          "reporting_status": "reported",
          "tag": "life-science"
        },
        {
          "count": 0,
          "coverage": "not-in-scope",
          "notes": "None of the twelve official research questions concerns protein design or optimization.",
          "reporting_status": "reported",
          "tag": "protein-design"
        },
        {
          "count": 0,
          "coverage": "not-in-scope",
          "notes": "None of the twelve official research questions concerns protein-protein binding.",
          "reporting_status": "reported",
          "tag": "protein-protein-binding"
        },
        {
          "count": 0,
          "coverage": "not-in-scope",
          "notes": "None of the twelve official research questions concerns protein-ligand binding.",
          "reporting_status": "reported",
          "tag": "protein-ligand-binding"
        },
        {
          "count": 0,
          "coverage": "not-in-scope",
          "notes": "The official question inventory contains no genomics or omics-analysis task.",
          "reporting_status": "reported",
          "tag": "genomics"
        },
        {
          "count": 0,
          "coverage": "not-in-scope",
          "notes": "The official question inventory contains no transcriptomics task.",
          "reporting_status": "reported",
          "tag": "transcriptomics"
        },
        {
          "count": 0,
          "coverage": "not-in-scope",
          "notes": "The official question inventory contains no single-cell task.",
          "reporting_status": "reported",
          "tag": "single-cell"
        }
      ],
      "domains": [
        "life-science"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "blade-evidence-identity",
          "locator": {
            "note": "Defines BLADE, the twelve source research questions/datasets, its decision-based purpose, and the two task types.",
            "type": "section",
            "value": "Title, abstract, and §§1–4"
          },
          "source_id": "blade-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/organizations",
            "/release_date",
            "/kind",
            "/capabilities",
            "/modalities",
            "/task_formats"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "blade-evidence-current-counts",
          "locator": {
            "note": "Reports 12 paired questions/datasets, 188 MCQs (20 conceptual-variable and 168 transform), and 536 ground-truth decisions (118 conceptual-variable, 246 transform, and 172 modeling).",
            "type": "section",
            "value": "arXiv v3 §§1, 4.1–4.2, 6 and Appendix A.2"
          },
          "source_id": "blade-arxiv-v3-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/versions/0/task_counts",
            "/versions/0/formal_tracks",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "blade-evidence-taxonomy",
          "locator": {
            "note": "AMTL, Crofoot, and Panda_nuts are labeled Evolutionary Biology; Fish is ecological/health-and-well-being. No official question is a protein, binding, genomics, transcriptomics, or single-cell task.",
            "type": "table",
            "value": "Table 3, complete twelve-question inventory"
          },
          "source_id": "blade-arxiv-v3-resource",
          "source_type": "resource",
          "supports": [
            "/domains",
            "/coverage_notes",
            "/access/biosafety_notes"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "blade-evidence-access-license",
          "locator": {
            "note": "Package 0.1.1; Apache-2.0 code; ODC-By-1.0 data; public task, annotation, baseline, and grader files.",
            "type": "repository-path",
            "value": "README.md, pyproject.toml, LICENSE, blade_bench/datasets/LICENSE, blade_bench/datasets/*, and blade_bench/eval/* at commit 6118fa8d5007b91aa8c91c518182db82446a4547"
          },
          "source_id": "blade-repository-resource",
          "source_type": "resource",
          "supports": [
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/implementations"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "blade-evidence-mcq-files",
          "locator": {
            "note": "The public files contain 20 conceptual-variable and 168 transformation MCQs across 11 source datasets, totaling 188.",
            "type": "repository-path",
            "value": "blade_bench/datasets/*/mcq_dataset.json and run_mcq.py at commit 6118fa8d5007b91aa8c91c518182db82446a4547"
          },
          "source_id": "blade-repository-resource",
          "source_type": "resource",
          "supports": [
            "/task_counts/subsets/0/count",
            "/task_counts/subsets/1/count",
            "/task_counts/subsets/2/count"
          ]
        }
      ],
      "field_status": [],
      "id": "blade",
      "implementations": [
        {
          "commit": "6118fa8d5007b91aa8c91c518182db82446a4547",
          "framework": "BLADE Python evaluator 0.1.1",
          "notes": "Official MCQ runner, one-turn and ReAct baselines, executable-analysis evaluator, prompts, and data.",
          "status": "official",
          "url": "https://github.com/behavioral-data/BLADE/tree/6118fa8d5007b91aa8c91c518182db82446a4547"
        }
      ],
      "kind": "suite",
      "latest_version": "arXiv v3",
      "modalities": [
        "text",
        "table",
        "code"
      ],
      "name": "BLADE",
      "organizations": [
        "University of Washington",
        "UC Berkeley",
        "New York University",
        "Stanford University",
        "University of British Columbia",
        "Microsoft",
        "George Washington University"
      ],
      "parent_id": null,
      "release_date": "2024-08-19",
      "resources": [
        {
          "access_notes": "Peer-reviewed Findings of EMNLP 2024 creator paper.",
          "id": "blade-paper-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY 4.0",
          "pin": null,
          "type": "paper",
          "url": "https://aclanthology.org/2024.findings-emnlp.815/"
        },
        {
          "access_notes": "Current creator manuscript, arXiv v3 dated 2025-11-10; the official source-archive response exposes the recorded SHA256 ETag.",
          "id": "blade-arxiv-v3-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY 4.0",
          "pin": {
            "kind": "snapshot",
            "url": "https://arxiv.org/src/2408.09667",
            "value": "sha256:8f37f27a5ca94540bf1eb8475c9dc838220a085788c402ba65fee980af807488"
          },
          "type": "paper",
          "url": "https://arxiv.org/abs/2408.09667"
        },
        {
          "access_notes": "Official project overview and result figures; captured HTML SHA256 eb90e6c54d830fdc2736c2e45d739359f455fc186a2845d880779a93ca8f554c.",
          "id": "blade-website-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "snapshot",
            "url": "https://blade-bench.github.io/",
            "value": "sha256:eb90e6c54d830fdc2736c2e45d739359f455fc186a2845d880779a93ca8f554c"
          },
          "type": "website",
          "url": "https://blade-bench.github.io/"
        },
        {
          "access_notes": "Official package version 0.1.1, benchmark data, prompts, baselines, and grader implementation; the repository has no release tags.",
          "id": "blade-repository-resource",
          "last_checked": "2026-07-21",
          "license": "Apache-2.0 for code; ODC-By-1.0 for data",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/behavioral-data/BLADE/tree/6118fa8d5007b91aa8c91c518182db82446a4547",
            "value": "6118fa8d5007b91aa8c91c518182db82446a4547"
          },
          "type": "repository",
          "url": "https://github.com/behavioral-data/BLADE"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "arXiv v3",
        "entries": [
          {
            "confidence": "high",
            "count": 12,
            "count_basis": "paired real-world research questions and datasets used as the source units for BLADE",
            "count_ref": "/task_counts/total",
            "count_unit": "problems",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "blade-evidence-current-counts"
            ],
            "mapping_method": "official-track",
            "notes": "Coverage is suite-wide; only four source questions are in life science.",
            "reporting_status": "reported",
            "task_type_id": "end-to-end-computational-analysis"
          }
        ],
        "notes": "BLADE is cross-domain; four source questions are biological or ecological.",
        "status": "partial"
      },
      "summary": "A cross-domain suite for discerning defensible analysis decisions and generating executable end-to-end analyses for open-ended scientific research questions; four of its twelve source questions are explicitly biological or ecological.",
      "task_counts": {
        "basis": "paired real-world research questions and datasets used as the source units for BLADE",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "individual decision-discrimination questions across the source datasets",
            "count": 188,
            "exclusive": false,
            "exhaustive": false,
            "id": "blade-mcq-items",
            "label": "Multiple-choice questions",
            "notes": "A separate question unit, not a partition of the 12 research-question/dataset pairs.",
            "reporting_status": "reported"
          },
          {
            "basis": "individual conceptual-variable discrimination questions",
            "count": 20,
            "exclusive": false,
            "exhaustive": false,
            "id": "blade-mcq-conceptual",
            "label": "Conceptual-variable MCQs",
            "notes": "One component of the 188-question MCQ track.",
            "reporting_status": "reported"
          },
          {
            "basis": "individual transformation-decision discrimination questions",
            "count": 168,
            "exclusive": false,
            "exhaustive": false,
            "id": "blade-mcq-transform",
            "label": "Transformation MCQs",
            "notes": "One component of the 188-question MCQ track.",
            "reporting_status": "reported"
          },
          {
            "basis": "unique defensible decisions in the expert ground-truth decision space",
            "count": 536,
            "exclusive": false,
            "exhaustive": false,
            "id": "blade-ground-truth-decisions",
            "label": "Ground-truth analysis decisions",
            "notes": "A grader-reference unit, not a count of benchmark tasks.",
            "reporting_status": "reported"
          },
          {
            "basis": "conceptual-variable decisions in the ground-truth decision space",
            "count": 118,
            "exclusive": false,
            "exhaustive": false,
            "id": "blade-ground-truth-conceptual",
            "label": "Ground-truth conceptual-variable decisions",
            "notes": "Together with transform and modeling decisions, partitions the 536 decision references rather than the 12 source tasks.",
            "reporting_status": "reported"
          },
          {
            "basis": "discrete data-transformation decisions in the ground-truth decision space",
            "count": 246,
            "exclusive": false,
            "exhaustive": false,
            "id": "blade-ground-truth-transform",
            "label": "Ground-truth transform decisions",
            "notes": "Together with conceptual-variable and modeling decisions, partitions the 536 decision references.",
            "reporting_status": "reported"
          },
          {
            "basis": "statistical-model and model-formula decisions in the ground-truth decision space",
            "count": 172,
            "exclusive": false,
            "exhaustive": false,
            "id": "blade-ground-truth-modeling",
            "label": "Ground-truth modeling decisions",
            "notes": "Together with conceptual-variable and transform decisions, partitions the 536 decision references.",
            "reporting_status": "reported"
          }
        ],
        "total": 12
      },
      "task_formats": [
        "multiple choice",
        "open-ended end-to-end scientific analysis"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Audited against the peer-reviewed paper, current arXiv v3, official project site, and commit-pinned official repository.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "blade-evidence-current-counts"
          ],
          "formal_tracks": [
            "blade-mcq",
            "blade-analysis-generation"
          ],
          "id": "blade-arxiv-v3",
          "label": "arXiv v3",
          "notes": "Current creator manuscript snapshot. It explicitly defines 188 MCQs and 536 ground-truth decisions across 12 source research-question/dataset pairs; the independently versioned code remains package 0.1.1 at the pinned commit.",
          "release_date": "2025-11-10",
          "status": "current",
          "task_counts": {
            "basis": "paired real-world research questions and datasets used as the source units for BLADE",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "individual decision-discrimination questions across the source datasets",
                "count": 188,
                "exclusive": false,
                "exhaustive": false,
                "id": "blade-mcq-items",
                "label": "Multiple-choice questions",
                "notes": "A separate question unit, not a partition of the 12 research-question/dataset pairs.",
                "reporting_status": "reported"
              },
              {
                "basis": "individual conceptual-variable discrimination questions",
                "count": 20,
                "exclusive": false,
                "exhaustive": false,
                "id": "blade-mcq-conceptual",
                "label": "Conceptual-variable MCQs",
                "notes": "One component of the 188-question MCQ track.",
                "reporting_status": "reported"
              },
              {
                "basis": "individual transformation-decision discrimination questions",
                "count": 168,
                "exclusive": false,
                "exhaustive": false,
                "id": "blade-mcq-transform",
                "label": "Transformation MCQs",
                "notes": "One component of the 188-question MCQ track.",
                "reporting_status": "reported"
              },
              {
                "basis": "unique defensible decisions in the expert ground-truth decision space",
                "count": 536,
                "exclusive": false,
                "exhaustive": false,
                "id": "blade-ground-truth-decisions",
                "label": "Ground-truth analysis decisions",
                "notes": "A grader-reference unit, not a count of benchmark tasks.",
                "reporting_status": "reported"
              },
              {
                "basis": "conceptual-variable decisions in the ground-truth decision space",
                "count": 118,
                "exclusive": false,
                "exhaustive": false,
                "id": "blade-ground-truth-conceptual",
                "label": "Ground-truth conceptual-variable decisions",
                "notes": "Together with transform and modeling decisions, partitions the 536 decision references rather than the 12 source tasks.",
                "reporting_status": "reported"
              },
              {
                "basis": "discrete data-transformation decisions in the ground-truth decision space",
                "count": 246,
                "exclusive": false,
                "exhaustive": false,
                "id": "blade-ground-truth-transform",
                "label": "Ground-truth transform decisions",
                "notes": "Together with conceptual-variable and modeling decisions, partitions the 536 decision references.",
                "reporting_status": "reported"
              },
              {
                "basis": "statistical-model and model-formula decisions in the ground-truth decision space",
                "count": 172,
                "exclusive": false,
                "exhaustive": false,
                "id": "blade-ground-truth-modeling",
                "label": "Ground-truth modeling decisions",
                "notes": "Together with conceptual-variable and transform decisions, partitions the 536 decision references.",
                "reporting_status": "reported"
              }
            ],
            "total": 12
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "Expert annotations and executable grader inputs are public under ODC-By-1.0.",
        "biosafety_notes": "Four tasks concern ecology, primate behavior, or evolutionary biology; none is a protein, omics, pathogen, intervention, or wet-lab task.",
        "grader": "Public hybrid evaluator combines code execution, value/graph matching, and GPT-4o-assisted semantic conversion and matching.",
        "level": "fully-open",
        "license": "Apache-2.0 for code; ODC-By-1.0 for data and annotations; CC BY 4.0 for the ACL paper",
        "tasks": "All twelve questions, data tables, schemas, and required submission shape are public."
      },
      "aliases": [
        "BLADE Task 2"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Task and decision counts, submission shape, access, grader, baseline settings, and metrics were verified.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "data-analysis",
        "coding",
        "tool-use",
        "scientific-reasoning"
      ],
      "coverage_notes": [
        {
          "count": 4,
          "coverage": "observed",
          "notes": "Four of the twelve official source questions are explicitly biological or ecological: AMTL, Crofoot, Panda_nuts, and Fish.",
          "reporting_status": "reported",
          "tag": "life-science"
        },
        {
          "count": 0,
          "coverage": "not-in-scope",
          "notes": "None of the twelve analysis-generation tasks concerns protein design.",
          "reporting_status": "reported",
          "tag": "protein-design"
        },
        {
          "count": 0,
          "coverage": "not-in-scope",
          "notes": "None of the twelve analysis-generation tasks concerns protein-protein binding.",
          "reporting_status": "reported",
          "tag": "protein-protein-binding"
        },
        {
          "count": 0,
          "coverage": "not-in-scope",
          "notes": "None of the twelve analysis-generation tasks concerns protein-ligand binding.",
          "reporting_status": "reported",
          "tag": "protein-ligand-binding"
        },
        {
          "count": 0,
          "coverage": "not-in-scope",
          "notes": "None of the twelve analysis-generation tasks is an omics-analysis task.",
          "reporting_status": "reported",
          "tag": "genomics"
        }
      ],
      "domains": [
        "life-science"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "blade-paper"
      ],
      "evaluation_run_ids": [
        "blade-creator-paper",
        "blade-creator-react"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "blade-generation-evidence-counts",
          "locator": {
            "note": "Defines twelve full-analysis tasks and 536 decision references: 118 conceptual-variable, 246 transform, and 172 modeling decisions.",
            "type": "section",
            "value": "arXiv v3 §§3–5 and Appendix A.2"
          },
          "source_id": "blade-generation-paper-resource",
          "source_type": "resource",
          "supports": [
            "/name",
            "/summary",
            "/kind",
            "/parent_id",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/versions/0/task_counts",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "blade-generation-evidence-implementation",
          "locator": {
            "note": "Public one-turn/ReAct runners, ten-step notebook sandbox, code execution, GPT-4o-assisted conversion/matching, structural matching, and result aggregation.",
            "type": "repository-path",
            "value": "run_gen_analyses.py, run_get_eval.py, run_scripts/, blade_bench/baselines/, and blade_bench/eval/ at commit 6118fa8d5007b91aa8c91c518182db82446a4547"
          },
          "source_id": "blade-generation-repository-resource",
          "source_type": "resource",
          "supports": [
            "/access",
            "/access/level",
            "/access/license",
            "/resources",
            "/implementations"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "blade-generation-evidence-taxonomy",
          "locator": {
            "note": "Four explicitly biological/ecological source questions and no protein, binding, or omics analysis task.",
            "type": "table",
            "value": "Table 3"
          },
          "source_id": "blade-generation-paper-resource",
          "source_type": "resource",
          "supports": [
            "/domains",
            "/coverage_notes",
            "/access/biosafety_notes"
          ]
        }
      ],
      "field_status": [],
      "id": "blade-analysis-generation",
      "implementations": [
        {
          "commit": "6118fa8d5007b91aa8c91c518182db82446a4547",
          "framework": "BLADE generation evaluator 0.1.1",
          "notes": "One-turn and ReAct generation, notebook sandbox, code execution, decision conversion, semantic and structural matching, and aggregation.",
          "status": "official",
          "url": "https://github.com/behavioral-data/BLADE/tree/6118fa8d5007b91aa8c91c518182db82446a4547"
        }
      ],
      "kind": "track",
      "latest_version": "arXiv v3",
      "modalities": [
        "text",
        "table",
        "code"
      ],
      "name": "BLADE End-to-End Analysis Generation",
      "organizations": [
        "University of Washington",
        "UC Berkeley",
        "New York University",
        "Stanford University",
        "University of British Columbia",
        "Microsoft",
        "George Washington University"
      ],
      "parent_id": "blade",
      "release_date": "2024-08-19",
      "resources": [
        {
          "access_notes": "Current v3 creator manuscript defining Task 2, ground-truth decision counts, metrics, and baselines.",
          "id": "blade-generation-paper-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY 4.0",
          "pin": {
            "kind": "snapshot",
            "url": "https://arxiv.org/src/2408.09667",
            "value": "sha256:8f37f27a5ca94540bf1eb8475c9dc838220a085788c402ba65fee980af807488"
          },
          "type": "paper",
          "url": "https://arxiv.org/abs/2408.09667"
        },
        {
          "access_notes": "Official prompts, one-turn and ReAct baselines, sandbox implementation, datasets, annotations, and evaluator.",
          "id": "blade-generation-repository-resource",
          "last_checked": "2026-07-21",
          "license": "Apache-2.0 for code; ODC-By-1.0 for data",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/behavioral-data/BLADE/tree/6118fa8d5007b91aa8c91c518182db82446a4547",
            "value": "6118fa8d5007b91aa8c91c518182db82446a4547"
          },
          "type": "repository",
          "url": "https://github.com/behavioral-data/BLADE"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "arXiv v3",
        "entries": [
          {
            "confidence": "high",
            "count": 12,
            "count_basis": "paired research questions and datasets requiring a complete generated analysis",
            "count_ref": "/task_counts/total",
            "count_unit": "problems",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "blade-generation-evidence-counts"
            ],
            "mapping_method": "official-track",
            "notes": "The track grades complete analysis decisions and execution.",
            "reporting_status": "reported",
            "task_type_id": "end-to-end-computational-analysis"
          }
        ],
        "notes": "Formal executable-analysis generation track.",
        "status": "complete"
      },
      "summary": "The BLADE track requiring a conceptual-variable specification, executable data-transformation function, and statistical-model function for each open-ended research question and dataset.",
      "task_counts": {
        "basis": "paired research questions and datasets requiring a complete generated analysis",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "unique defensible decisions used by the grader across all source tasks",
            "count": 536,
            "exclusive": false,
            "exhaustive": false,
            "id": "blade-generation-ground-truth-decisions",
            "label": "Ground-truth analysis decisions",
            "notes": "A grader-reference unit, not a partition of the 12 generated-analysis tasks.",
            "reporting_status": "reported"
          },
          {
            "basis": "conceptual-variable decisions in the expert ground truth",
            "count": 118,
            "exclusive": false,
            "exhaustive": false,
            "id": "blade-generation-conceptual-decisions",
            "label": "Conceptual-variable decisions",
            "notes": "A decision-reference count.",
            "reporting_status": "reported"
          },
          {
            "basis": "discrete transformation decisions in the expert ground truth",
            "count": 246,
            "exclusive": false,
            "exhaustive": false,
            "id": "blade-generation-transform-decisions",
            "label": "Transform decisions",
            "notes": "A decision-reference count.",
            "reporting_status": "reported"
          },
          {
            "basis": "statistical-model and formula decisions in the expert ground truth",
            "count": 172,
            "exclusive": false,
            "exhaustive": false,
            "id": "blade-generation-modeling-decisions",
            "label": "Modeling decisions",
            "notes": "A decision-reference count.",
            "reporting_status": "reported"
          }
        ],
        "total": 12
      },
      "task_formats": [
        "open-ended end-to-end scientific analysis"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Audited against creator paper v3 and the commit-pinned official generation and evaluation code.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "blade-generation-evidence-counts"
          ],
          "formal_tracks": [],
          "id": "blade-generation-arxiv-v3",
          "label": "arXiv v3",
          "notes": "Current creator-manuscript task and decision-space snapshot.",
          "release_date": "2025-11-10",
          "status": "current",
          "task_counts": {
            "basis": "paired research questions and datasets requiring a complete generated analysis",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "unique defensible decisions used by the grader across all source tasks",
                "count": 536,
                "exclusive": false,
                "exhaustive": false,
                "id": "blade-generation-ground-truth-decisions",
                "label": "Ground-truth analysis decisions",
                "notes": "A grader-reference unit, not a partition of the 12 generated-analysis tasks.",
                "reporting_status": "reported"
              },
              {
                "basis": "conceptual-variable decisions in the expert ground truth",
                "count": 118,
                "exclusive": false,
                "exhaustive": false,
                "id": "blade-generation-conceptual-decisions",
                "label": "Conceptual-variable decisions",
                "notes": "A decision-reference count.",
                "reporting_status": "reported"
              },
              {
                "basis": "discrete transformation decisions in the expert ground truth",
                "count": 246,
                "exclusive": false,
                "exhaustive": false,
                "id": "blade-generation-transform-decisions",
                "label": "Transform decisions",
                "notes": "A decision-reference count.",
                "reporting_status": "reported"
              },
              {
                "basis": "statistical-model and formula decisions in the expert ground truth",
                "count": 172,
                "exclusive": false,
                "exhaustive": false,
                "id": "blade-generation-modeling-decisions",
                "label": "Modeling decisions",
                "notes": "A decision-reference count.",
                "reporting_status": "reported"
              }
            ],
            "total": 12
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "Public dataset descriptions and MCQ records; benchmark data are ODC-By-1.0.",
        "biosafety_notes": "The track contains no protein, omics, pathogen, intervention, or wet-lab task.",
        "grader": "Public deterministic exact-choice accuracy implementation.",
        "level": "fully-open",
        "license": "Apache-2.0 for code; ODC-By-1.0 for data and MCQ records; CC BY 4.0 for the ACL paper",
        "tasks": "All 188 MCQ objects and their choices are public in eleven commit-pinned dataset directories."
      },
      "aliases": [
        "BLADE Task 1"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Counts, task definition, prompt, metric, access, and official protocol were verified.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "knowledge",
        "classification",
        "data-analysis",
        "scientific-reasoning"
      ],
      "coverage_notes": [
        {
          "count": null,
          "coverage": "observed",
          "notes": "Biological source datasets are present, but the paper does not publish an MCQ count by scientific domain.",
          "reporting_status": "not_reported",
          "tag": "life-science"
        },
        {
          "count": 0,
          "coverage": "not-in-scope",
          "notes": "No official BLADE source research question concerns protein design.",
          "reporting_status": "reported",
          "tag": "protein-design"
        },
        {
          "count": 0,
          "coverage": "not-in-scope",
          "notes": "No official BLADE source research question concerns protein-protein binding.",
          "reporting_status": "reported",
          "tag": "protein-protein-binding"
        },
        {
          "count": 0,
          "coverage": "not-in-scope",
          "notes": "No official BLADE source research question concerns protein-ligand binding.",
          "reporting_status": "reported",
          "tag": "protein-ligand-binding"
        }
      ],
      "domains": [
        "life-science"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "blade-paper"
      ],
      "evaluation_run_ids": [
        "blade-creator-decision-mcq"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "blade-mcq-evidence-counts",
          "locator": {
            "note": "Defines the MCQ task and reports 188 questions: 20 conceptual-variable and 168 transformation questions.",
            "type": "section",
            "value": "arXiv v3 §4.1, §6, Figure 3, and Appendix A.6"
          },
          "source_id": "blade-mcq-paper-resource",
          "source_type": "resource",
          "supports": [
            "/name",
            "/summary",
            "/kind",
            "/parent_id",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/versions/0/task_counts",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "blade-mcq-evidence-files",
          "locator": {
            "note": "Public files reproduce 20/168/188 across eleven source datasets and implement a single prompt call plus exact-choice accuracy.",
            "type": "repository-path",
            "value": "blade_bench/datasets/*/mcq_dataset.json, blade_bench/baselines/lm/mcq.py, run_mcq.py, and blade_bench/eval/datamodel/run_mcq.py at commit 6118fa8d5007b91aa8c91c518182db82446a4547"
          },
          "source_id": "blade-mcq-repository-resource",
          "source_type": "resource",
          "supports": [
            "/task_counts",
            "/access",
            "/access/level",
            "/access/license",
            "/resources",
            "/implementations"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "blade-mcq-evidence-taxonomy",
          "locator": {
            "note": "The source-question inventory includes biological/ecological datasets but no protein, binding, or omics question; MCQ-by-domain counts are not reported.",
            "type": "table",
            "value": "Table 3"
          },
          "source_id": "blade-mcq-paper-resource",
          "source_type": "resource",
          "supports": [
            "/domains",
            "/coverage_notes",
            "/access/biosafety_notes"
          ]
        }
      ],
      "field_status": [],
      "id": "blade-mcq",
      "implementations": [
        {
          "commit": "6118fa8d5007b91aa8c91c518182db82446a4547",
          "framework": "BLADE MCQ runner 0.1.1",
          "notes": "run_mcq.py plus public prompt, MCQ records, and exact-choice metrics.",
          "status": "official",
          "url": "https://github.com/behavioral-data/BLADE/tree/6118fa8d5007b91aa8c91c518182db82446a4547"
        }
      ],
      "kind": "track",
      "latest_version": "arXiv v3",
      "modalities": [
        "text",
        "table"
      ],
      "name": "BLADE Decision-Discrimination MCQ",
      "organizations": [
        "University of Washington",
        "UC Berkeley",
        "New York University",
        "Stanford University",
        "University of British Columbia",
        "Microsoft",
        "George Washington University"
      ],
      "parent_id": "blade",
      "release_date": "2024-08-19",
      "resources": [
        {
          "access_notes": "Current v3 creator manuscript defining Task 1 and its counts.",
          "id": "blade-mcq-paper-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY 4.0",
          "pin": {
            "kind": "snapshot",
            "url": "https://arxiv.org/src/2408.09667",
            "value": "sha256:8f37f27a5ca94540bf1eb8475c9dc838220a085788c402ba65fee980af807488"
          },
          "type": "paper",
          "url": "https://arxiv.org/abs/2408.09667"
        },
        {
          "access_notes": "Official MCQ data, prompt, runner, and deterministic metrics.",
          "id": "blade-mcq-repository-resource",
          "last_checked": "2026-07-21",
          "license": "Apache-2.0 for code; ODC-By-1.0 for data",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/behavioral-data/BLADE/tree/6118fa8d5007b91aa8c91c518182db82446a4547",
            "value": "6118fa8d5007b91aa8c91c518182db82446a4547"
          },
          "type": "repository",
          "url": "https://github.com/behavioral-data/BLADE"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "arXiv v3",
        "entries": [
          {
            "confidence": "high",
            "count": 188,
            "count_basis": "individual multiple-choice decision-discrimination questions",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "blade-mcq-evidence-counts"
            ],
            "mapping_method": "official-track",
            "notes": "Questions test defensibility of scientific analysis decisions.",
            "reporting_status": "reported",
            "task_type_id": "scientific-evidence-interpretation"
          }
        ],
        "notes": "Formal decision-discrimination track.",
        "status": "complete"
      },
      "summary": "The BLADE track for selecting the most or least justifiable conceptual-variable and data-transformation decisions for a research question and dataset.",
      "task_counts": {
        "basis": "individual multiple-choice decision-discrimination questions",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "individual questions asking which conceptual variable is most or least justifiable",
            "count": 20,
            "exclusive": true,
            "exhaustive": true,
            "id": "blade-mcq-conceptual-items",
            "label": "Conceptual-variable MCQs",
            "notes": "Mutually exclusive with transformation MCQs.",
            "reporting_status": "reported"
          },
          {
            "basis": "individual questions asking which transformation is most or least justifiable",
            "count": 168,
            "exclusive": true,
            "exhaustive": true,
            "id": "blade-mcq-transform-items",
            "label": "Transformation MCQs",
            "notes": "Mutually exclusive with conceptual-variable MCQs.",
            "reporting_status": "reported"
          },
          {
            "basis": "official dataset directories containing mcq_dataset.json at the pinned commit",
            "count": 11,
            "exclusive": false,
            "exhaustive": false,
            "id": "blade-mcq-source-datasets",
            "label": "Source datasets represented",
            "notes": "A source-dataset count, not a partition of questions; the soccer dataset is part of analysis generation but has no MCQ file.",
            "reporting_status": "reported"
          }
        ],
        "total": 188
      },
      "task_formats": [
        "multiple choice"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Audited against creator paper v3 and the commit-pinned official MCQ files and runner.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "blade-mcq-evidence-counts"
          ],
          "formal_tracks": [],
          "id": "blade-mcq-arxiv-v3",
          "label": "arXiv v3",
          "notes": "Current published MCQ inventory and task definition.",
          "release_date": "2025-11-10",
          "status": "current",
          "task_counts": {
            "basis": "individual multiple-choice decision-discrimination questions",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "individual questions asking which conceptual variable is most or least justifiable",
                "count": 20,
                "exclusive": true,
                "exhaustive": true,
                "id": "blade-mcq-conceptual-items",
                "label": "Conceptual-variable MCQs",
                "notes": "Mutually exclusive with transformation MCQs.",
                "reporting_status": "reported"
              },
              {
                "basis": "individual questions asking which transformation is most or least justifiable",
                "count": 168,
                "exclusive": true,
                "exhaustive": true,
                "id": "blade-mcq-transform-items",
                "label": "Transformation MCQs",
                "notes": "Mutually exclusive with conceptual-variable MCQs.",
                "reporting_status": "reported"
              },
              {
                "basis": "official dataset directories containing mcq_dataset.json at the pinned commit",
                "count": 11,
                "exclusive": false,
                "exhaustive": false,
                "id": "blade-mcq-source-datasets",
                "label": "Source datasets represented",
                "notes": "A source-dataset count, not a partition of questions; the soccer dataset is part of analysis generation but has no MCQ file.",
                "reporting_status": "reported"
              }
            ],
            "total": 188
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "See the linked official creator resources.",
        "biosafety_notes": null,
        "grader": "Not established by the independently verified metadata claims.",
        "level": "fully-open",
        "license": null,
        "tasks": "See the linked official creator resources."
      },
      "aliases": [],
      "audit": {
        "audited_date": "2026-07-31",
        "notes": "Automated double-pass extraction plus deterministic official-resource pin; the benchmark license remains unverified and is published as null.",
        "status": "audited-with-caveats",
        "unresolved_fields": 2
      },
      "benchmark_use_ids": [
        "an-integrative-approach-to-protein-sequence-design-thr-cam-benchmark-5-use",
        "an-integrative-approach-to-protein-sequence-design-thr-cam-benchmark-6-use"
      ],
      "capabilities": [
        "design",
        "optimization"
      ],
      "coverage_notes": [],
      "domains": [
        "life-science",
        "protein-science",
        "protein-sequence",
        "protein-structure",
        "protein-design",
        "protein-protein-binding"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "an-integrative-approach-to-protein-sequence-design-thr"
      ],
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-31",
          "id": "cam-benchmark-automated-metadata-1-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "cbdee6951e5532b57429398b51436fe8e2abfc9f78df018593e82380271ae3fc",
            "type": "section",
            "value": "Data Availability"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/access/level"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "cam-benchmark-automated-metadata-2-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "fb77bf96dd1177ced1e894036d1f38021a551bc2d2586536f0957b1fe6c1c5c7",
            "type": "section",
            "value": "Results — Genetic algorithms can be applied to higher-dimensional design problems"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/capabilities",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "cam-benchmark-automated-metadata-3-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "c2e6ebd692eec377963d0801204fd440900ed82562f2b8efbdc563b59a8f0b27",
            "type": "section",
            "value": "Results — The random resetting mutation operator results in slow convergence; Genetic algorithms can be applied to higher-dimensional design problems"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/domains"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "cam-benchmark-automated-metadata-4-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "6e6903443229eedd52871ff2829f1ae65202e38c28b6be6dcf62d0c145ae0384",
            "type": "section",
            "value": "Data Availability"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/kind"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "cam-benchmark-automated-metadata-5-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "d8c46376609c5c99b98d4c377c8063ef215856243b1c93db16fcae9f8b8d62c1",
            "type": "figure",
            "value": "Fig. 1"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/modalities"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "cam-benchmark-automated-metadata-6-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "87b872e188012dde7bc79f0050ec56e04f3529823616351d129b2015f528a74b",
            "type": "section",
            "value": "Methods — Structure preparation"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/name"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "cam-benchmark-automated-metadata-7-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "06c4d9bdb31c2ac3c4e3ceb12dff9a706e3bc8b0d242cbef9e52e7f81cb2a18e",
            "type": "other",
            "value": "Front matter — author-affiliation mapping"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/organizations"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "cam-benchmark-automated-metadata-8-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "cb6a6e74c8710e5957b6fad88effc7986b217bb0d4d48cbddd3f14020c688d37",
            "type": "section",
            "value": "Results — Genetic algorithms can be applied to higher-dimensional design problems"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/summary"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "cam-benchmark-automated-bibliographic-evidence",
          "locator": {
            "document_page": null,
            "note": "Resolved from the canonical paper identifier during intake.",
            "printed_page": null,
            "source_fragment_sha256": "5b45963847a151050eb2419c23684ca38e9325e2457145144b7ec0d2ea01d410",
            "type": "other",
            "value": "Crossref bibliographic metadata"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/release_date"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "cam-benchmark-automated-count-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "4da017980304147a6b12b267fc5e04e7d36f9c5f339976603d592f37892c0c4a",
            "type": "figure",
            "value": "Fig. 1"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/task_counts/total",
            "/task_counts/basis",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "cam-benchmark-automated-subset-coverage-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "4da017980304147a6b12b267fc5e04e7d36f9c5f339976603d592f37892c0c4a",
            "type": "figure",
            "value": "Fig. 1"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/task_counts/subsets",
            "/versions/0/task_counts/subsets"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "cam-benchmark-automated-resource-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "cbdee6951e5532b57429398b51436fe8e2abfc9f78df018593e82380271ae3fc",
            "type": "section",
            "value": "Data Availability"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/resources",
            "/implementations",
            "/access/license"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "cam-benchmark-automated-creator-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "6115205069f1a4999f060235c61c133f276780408241486c68044ff841a14049",
            "type": "other",
            "value": "Front matter — DOI"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/resources/0"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "cam-benchmark-automated-version-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "6115205069f1a4999f060235c61c133f276780408241486c68044ff841a14049",
            "type": "other",
            "value": "Front matter — DOI"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/latest_version",
            "/versions/0"
          ]
        }
      ],
      "field_status": [
        {
          "confidence": "high",
          "evidence_ids": [
            "cam-benchmark-automated-subset-coverage-evidence"
          ],
          "path": "/task_counts/subsets",
          "reason": "The double-pass review established the root item total but did not establish an exhaustive formal-subset inventory; the empty list must not be interpreted as evidence that the benchmark has no subsets.",
          "status": "provisional"
        },
        {
          "confidence": "high",
          "evidence_ids": [
            "cam-benchmark-automated-resource-evidence"
          ],
          "path": "/access/license",
          "reason": "The double-pass review verified the official resource identity but did not establish a redistributable benchmark license; the value remains null pending source-level license verification.",
          "status": "provisional"
        }
      ],
      "id": "cam-benchmark",
      "implementations": [
        {
          "commit": "b16a0ef3d5c44f65714b1a6c51826f6b4bdaa998",
          "framework": "official repository",
          "notes": "Commit resolved deterministically during paper intake.",
          "status": "official",
          "url": "https://github.com/luhong88/int_seq_des"
        }
      ],
      "kind": "dataset",
      "latest_version": "initial-release",
      "modalities": [
        "protein-sequence",
        "structure-3d"
      ],
      "name": "CaM benchmark",
      "organizations": [
        "University of California, San Francisco",
        "Quantitative Biosciences Institute",
        "Chan Zuckerberg Biohub"
      ],
      "parent_id": null,
      "release_date": "2024-07-11",
      "resources": [
        {
          "access_notes": "Versioned creator source; full text is not mirrored.",
          "id": "cam-benchmark-creator-paper-resource",
          "last_checked": "2026-07-31",
          "license": null,
          "pin": null,
          "type": "paper",
          "url": "https://doi.org/10.1371/journal.pcbi.1011953"
        },
        {
          "access_notes": "Official resource pinned immutably during intake.",
          "id": "cam-benchmark-official-repository-resource",
          "last_checked": "2026-07-31",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://github.com/luhong88/int_seq_des/commit/b16a0ef3d5c44f65714b1a6c51826f6b4bdaa998",
            "value": "b16a0ef3d5c44f65714b1a6c51826f6b4bdaa998"
          },
          "type": "repository",
          "url": "https://github.com/luhong88/int_seq_des"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "initial-release",
        "entries": [
          {
            "confidence": "high",
            "count": null,
            "count_basis": "Creator-defined multistate protein sequence-design benchmark system.",
            "count_ref": null,
            "count_unit": "systems",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "cam-benchmark-automated-metadata-2-evidence"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "The fourteen modeled structural states are objective dimensions, not fourteen independent sequence-design tasks.",
            "reporting_status": "not_reported",
            "task_type_id": "protein-sequence-design"
          }
        ],
        "notes": "The creator paper explicitly frames this benchmark system as multistate protein sequence design; more specific downstream design objectives are not exhaustively classified.",
        "status": "partial"
      },
      "summary": "A multistate protein sequence-design benchmark spanning CaM conformations and binding modes.",
      "task_counts": {
        "basis": "Fig. 1B explicitly defines the complete CaM design problem as fourteen-state.",
        "reporting_status": "reported",
        "subsets": [],
        "total": 14
      },
      "task_formats": [
        "unclassified"
      ],
      "verification": {
        "last_verified": "2026-07-31",
        "notes": "New family admitted after creator source and official resource verification; the unresolved benchmark license is visibly flagged.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": null,
          "evidence_ids": [
            "cam-benchmark-automated-count-evidence",
            "cam-benchmark-automated-version-evidence"
          ],
          "formal_tracks": [],
          "id": "cam-benchmark-initial-release-version",
          "label": "initial-release",
          "notes": "Atlas snapshot label for the creator-paper release; the source does not report a formal benchmark version.",
          "release_date": "2024-07-11",
          "status": "current",
          "task_counts": {
            "basis": "Fig. 1B explicitly defines the complete CaM design problem as fourteen-state.",
            "reporting_status": "reported",
            "subsets": [],
            "total": 14
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "Public-server targets, model-1 results, reference structures, predictions, scores, and time-window downloads are public after release; models 2-5 and advanced scores are in downloads.",
        "biosafety_notes": "The registry mirrors no pre-release sequences, structures, ligands, predictions, or development-server results.",
        "grader": "Fully automated OpenStructure scoring against released PDB biounits; no single metric or overall ranking is defined.",
        "level": "partially-open",
        "license": "CAMEO-provided data and submitted redistributable predictions are CC BY-SA 4.0; AlphaFold 3 outputs retain separate terms.",
        "tasks": "Public and development servers receive selected PDB pre-release entries; participation requires server registration and development-server names remain disguised."
      },
      "aliases": [
        "Continuous Automated Model EvaluatiOn"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "The current scope and 2024 bounded study are verified; the exact first service day in 2012 is not reported.",
        "status": "audited-with-caveats",
        "unresolved_fields": 1
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction"
      ],
      "coverage_notes": [
        {
          "count": null,
          "coverage": "explicitly-in-scope",
          "notes": "Medium and hard complete-complex targets include homomers, heteromers, and antibodies; the living service has no fixed all-time count.",
          "reporting_status": "not_reported",
          "tag": "protein-protein-binding"
        },
        {
          "count": null,
          "coverage": "explicitly-in-scope",
          "notes": "Ligand targets and ligand-containing medium/hard targets are in scope; the living service has no fixed all-time count.",
          "reporting_status": "not_reported",
          "tag": "protein-ligand-binding"
        },
        {
          "count": null,
          "coverage": "explicitly-in-scope",
          "notes": "Antibody complexes are analyzed as a protein-protein subclass; no current rolling count is asserted.",
          "reporting_status": "not_reported",
          "tag": "antibody-antigen"
        }
      ],
      "domains": [
        "protein-structure",
        "protein-protein-binding",
        "protein-ligand-binding",
        "antibody-antigen"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "cameo-paper"
      ],
      "evaluation_run_ids": [
        "cameo-2024-antibody-three-server-common",
        "cameo-2024-ligand-baseline-common",
        "cameo-2024-ppi-three-server-common"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "cameo-evidence-origin",
          "locator": {
            "note": "Documents the creator-operated continuous platform and its weekly blind PDB pre-release workflow since 2012.",
            "type": "section",
            "value": "Abstract and Introduction"
          },
          "source_id": "cameo-foundation-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/kind",
            "/organizations",
            "/release_date"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "cameo-evidence-legacy",
          "locator": {
            "note": "Identifies the historical categories and the April 2025 end of single-chain 3D.",
            "type": "section",
            "value": "Archive notice and General Workflow"
          },
          "source_id": "cameo-legacy-help-resource",
          "source_type": "resource",
          "supports": [
            "/versions/0/as_of",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "cameo-evidence-2024-study",
          "locator": {
            "note": "Reports the 2024 window, 14,078 PDB releases, 7,150 selected targets, category counts, common subsets, models, metrics, and public downloads.",
            "type": "section",
            "value": "Sections 2.1-2.5 and 3; Figures 1-4"
          },
          "source_id": "cameo-paper",
          "source_type": "work",
          "supports": [
            "/summary",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/coverage_notes",
            "/versions/1/release_date",
            "/versions/1/as_of",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets",
            "/resources",
            "/implementations",
            "/scientific_task_classification/entries/1",
            "/scientific_task_classification/entries/2"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "cameo-evidence-current-help",
          "locator": {
            "note": "Defines the current complex-only rolling category, target unit, four-day window, up-to-five models, OpenStructure scoring, common-subset aggregation, and access.",
            "type": "section",
            "value": "General Workflow; Target Submission; Prediction Format; Evaluation; Scores aggregation"
          },
          "source_id": "cameo-current-help-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/biosafety_notes",
            "/versions/2/as_of",
            "/versions/2/task_counts/total",
            "/versions/2/task_counts/basis",
            "/versions/2/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "cameo-evidence-terms",
          "locator": {
            "note": "Public versus development-server access and CC BY-SA 4.0 redistribution terms.",
            "type": "section",
            "value": "Description; Terms; Copyright"
          },
          "source_id": "cameo-terms-resource",
          "source_type": "resource",
          "supports": [
            "/access/level",
            "/access/license"
          ]
        }
      ],
      "field_status": [
        {
          "confidence": "medium",
          "evidence_ids": [
            "cameo-evidence-origin"
          ],
          "path": "/release_date",
          "reason": "Creator sources establish that CAMEO has operated since 2012 but do not report an exact first service day; 2012-01-01 is a year-normalized date.",
          "status": "provisional"
        }
      ],
      "id": "cameo",
      "implementations": [
        {
          "commit": null,
          "framework": "CAMEO hosted evaluation service using OpenStructure",
          "notes": "The hosted grader is operational and documented; no public commit-pinned deployment source is asserted.",
          "status": "official",
          "url": "https://cameo3d.org/"
        }
      ],
      "kind": "competition",
      "latest_version": "current-complex-3d",
      "modalities": [
        "protein-sequence",
        "dna-rna-sequence",
        "structure-3d"
      ],
      "name": "CAMEO",
      "organizations": [
        "SIB Swiss Institute of Bioinformatics",
        "Biozentrum University of Basel"
      ],
      "parent_id": null,
      "release_date": "2012-01-01",
      "resources": [
        {
          "access_notes": "Current official result and comparison interface.",
          "id": "cameo-current-site-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY-SA 4.0 for CAMEO-provided data",
          "pin": {
            "kind": "snapshot",
            "url": "https://cameo3d.org/",
            "value": "2026-07-21"
          },
          "type": "website",
          "url": "https://cameo3d.org/"
        },
        {
          "access_notes": "Current workflow, filtering, formats, metrics, aggregation, and download documentation.",
          "id": "cameo-current-help-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY-SA 4.0 for CAMEO-provided data",
          "pin": {
            "kind": "snapshot",
            "url": "https://cameo3d.org/help",
            "value": "2026-07-21"
          },
          "type": "documentation",
          "url": "https://cameo3d.org/help"
        },
        {
          "access_notes": "Official access and redistribution terms.",
          "id": "cameo-terms-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY-SA 4.0",
          "pin": {
            "kind": "snapshot",
            "url": "https://cameo3d.org/terms",
            "value": "2026-07-21"
          },
          "type": "documentation",
          "url": "https://cameo3d.org/terms"
        },
        {
          "access_notes": "Final creator paper for the complete-complex platform and 2024 study snapshot.",
          "id": "cameo-complex-paper-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY 4.0",
          "pin": {
            "kind": "version",
            "url": "https://doi.org/10.1002/prot.70060",
            "value": "Proteins 94(1):403-413"
          },
          "type": "paper",
          "url": "https://doi.org/10.1002/prot.70060"
        },
        {
          "access_notes": "Creator paper documenting the original continuously running platform.",
          "id": "cameo-foundation-paper-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "version",
            "url": "https://doi.org/10.1002/prot.25431",
            "value": "Proteins 86(S1):387-398"
          },
          "type": "paper",
          "url": "https://doi.org/10.1002/prot.25431"
        },
        {
          "access_notes": "Official archive of the discontinued single-chain 3D category, which ended in April 2025.",
          "id": "cameo-legacy-help-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY-SA 4.0 for CAMEO-provided data",
          "pin": {
            "kind": "snapshot",
            "url": "https://archive.cameo3d.org/cameong_help/3d/",
            "value": "legacy-single-chain-ended-2025-04"
          },
          "type": "documentation",
          "url": "https://archive.cameo3d.org/cameong_help/3d/"
        }
      ],
      "scientific_task_classification": {
        "as_of": "2026-07-21",
        "benchmark_version": "current-complex-3d",
        "entries": [
          {
            "confidence": "high",
            "count": null,
            "count_basis": "Weekly complete-complex targets in the current CAMEO service.",
            "count_ref": null,
            "count_unit": "targets",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "cameo-evidence-current-help"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "No fixed all-time count exists for the rolling service.",
            "reporting_status": "not_reported",
            "task_type_id": "protein-complex-structure-prediction"
          },
          {
            "confidence": "high",
            "count": null,
            "count_basis": "Weekly targets and submitted server models.",
            "count_ref": null,
            "count_unit": "targets",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "cameo-evidence-2024-study"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "CAMEO evaluates prediction quality against newly released structures.",
            "reporting_status": "not_reported",
            "task_type_id": "protein-model-quality-assessment"
          },
          {
            "confidence": "high",
            "count": null,
            "count_basis": "Ligand-containing targets in the current rolling service.",
            "count_ref": null,
            "count_unit": "targets",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "cameo-evidence-2024-study"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "Ligand pose is in scope; binding affinity is not asserted.",
            "reporting_status": "not_reported",
            "task_type_id": "protein-ligand-pose-prediction"
          }
        ],
        "notes": "Rolling service; task mappings describe current public categories and never imply an all-time target total.",
        "status": "partial"
      },
      "summary": "Weekly, automated, independent, blind evaluation of registered macromolecular structure-prediction servers on complete PDB entries whose experimental structures are withheld during prediction.",
      "task_counts": {
        "basis": "selected complete PDB entries accumulated by an unbounded weekly rolling service",
        "reporting_status": "not_reported",
        "subsets": [],
        "total": null
      },
      "task_formats": [
        "continuous blind complete-complex structure prediction",
        "stoichiometry prediction",
        "ligand pose prediction"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Audited against final creator papers and the current official workflow, terms, and archived category documentation.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2025-04-05",
          "evidence_ids": [
            "cameo-evidence-origin",
            "cameo-evidence-legacy"
          ],
          "formal_tracks": [],
          "id": "cameo-legacy-multicategory",
          "label": "legacy-multicategory",
          "notes": "The archive states that the old single-chain 3D category ended in April 2025; QE, CP, and LB are also discontinued.",
          "release_date": "2012-01-01",
          "status": "superseded",
          "task_counts": {
            "basis": "rolling targets across now-discontinued single-chain 3D, quality-estimation, contact-prediction, and ligand-binding-site categories",
            "reporting_status": "not_reported",
            "subsets": [],
            "total": null
          }
        },
        {
          "as_of": "2024-11-30",
          "evidence_ids": [
            "cameo-evidence-2024-study"
          ],
          "formal_tracks": [],
          "id": "cameo-2024-complex-study",
          "label": "2024-complex-study",
          "notes": "A bounded creator-paper snapshot; it is not an all-time CAMEO total.",
          "release_date": "2024-01-06",
          "status": "superseded",
          "task_counts": {
            "basis": "selected interesting complete PDB entries in the creator-paper study window",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "selected PDB entries",
                "count": 1332,
                "exclusive": true,
                "exhaustive": true,
                "id": "cameo-2024-medium",
                "label": "Medium complete-complex targets",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "selected PDB entries",
                "count": 1981,
                "exclusive": true,
                "exhaustive": true,
                "id": "cameo-2024-hard",
                "label": "Hard complete-complex targets",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "selected PDB entries",
                "count": 3837,
                "exclusive": true,
                "exhaustive": true,
                "id": "cameo-2024-ligand",
                "label": "Ligand complete-complex targets",
                "notes": "Easy complexes containing a ligand novel relative to their easy templates.",
                "reporting_status": "reported"
              },
              {
                "basis": "targets",
                "count": 3615,
                "exclusive": false,
                "exhaustive": false,
                "id": "cameo-2024-ligand-plinder-mapped",
                "label": "Ligand targets mapped to PLINDER",
                "notes": "Corresponds to 9,459 ligands after excluding 222 targets from the mapping analysis.",
                "reporting_status": "reported"
              },
              {
                "basis": "targets predicted by all four baseline servers",
                "count": 2584,
                "exclusive": false,
                "exhaustive": false,
                "id": "cameo-2024-ligand-baseline-common",
                "label": "Ligand baseline common subset",
                "notes": "Contains 6,152 non-polymer entities.",
                "reporting_status": "reported"
              },
              {
                "basis": "medium/hard protein-only targets",
                "count": 2078,
                "exclusive": false,
                "exhaustive": false,
                "id": "cameo-2024-ppi-after-monomer-exclusion",
                "label": "Protein-protein targets after monomer exclusion",
                "notes": "Derived from 2,775 medium/hard targets after excluding 697 monomers.",
                "reporting_status": "reported"
              },
              {
                "basis": "protein-protein targets mapped to SAbDab",
                "count": 525,
                "exclusive": false,
                "exhaustive": false,
                "id": "cameo-2024-antibody-candidates",
                "label": "Antibody targets",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "protein-protein targets",
                "count": 836,
                "exclusive": false,
                "exhaustive": false,
                "id": "cameo-2024-homomer-candidates",
                "label": "Homomer targets",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "protein-protein targets",
                "count": 717,
                "exclusive": false,
                "exhaustive": false,
                "id": "cameo-2024-heteromer-candidates",
                "label": "Non-antibody heteromer targets",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "targets predicted by AlphaFold 3, MultiFOLD, and SWISS-MODEL",
                "count": 392,
                "exclusive": false,
                "exhaustive": false,
                "id": "cameo-2024-ppi-three-server-common",
                "label": "Protein-protein three-server common subset",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "common-subset targets",
                "count": 83,
                "exclusive": false,
                "exhaustive": false,
                "id": "cameo-2024-antibody-three-server-common",
                "label": "Antibody targets in the three-server common subset",
                "notes": "The paper reports that 61 of these 83 antibody targets have one protein chain per entity.",
                "reporting_status": "reported"
              }
            ],
            "total": 7150
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "cameo-evidence-current-help"
          ],
          "formal_tracks": [],
          "id": "cameo-current-complex-3d",
          "label": "current-complex-3d",
          "notes": "Current single category: complex structures modeling (3D). Counts change weekly and no fixed all-time snapshot is asserted.",
          "release_date": null,
          "status": "rolling",
          "task_counts": {
            "basis": "selected complete PDB entries accumulated by an unbounded weekly rolling service",
            "reporting_status": "not_reported",
            "subsets": [],
            "total": null
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "Completed-round targets, submitted predictions, numerical score tables, rankings, presentations, and round archives are publicly linked; some source structures or pharmaceutical data may retain provider-specific terms.",
        "biosafety_notes": "The registry mirrors no target sequences, structures, ligand sets, or submitted models; active targets can include viral, immune, or drug-design systems.",
        "grader": "Category-specific numerical pipelines and independent assessor teams; no single round-independent grader or metric exists.",
        "level": "partially-open",
        "license": "CC BY 4.0 for Prediction Center website content; linked target and prediction artifacts may have separate terms",
        "tasks": "Target sequences and submission metadata are public during a round; experimental structures are intentionally withheld until prediction closes."
      },
      "aliases": [
        "Critical Assessment of Structure Prediction",
        "Critical Assessment of Techniques for Protein Structure Prediction"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "CASP17 is explicitly represented as a dated rolling round and CASP16 as the latest completed round; count units distinguish target releases, unique assessed targets, and evaluation units.",
        "status": "audited-with-caveats",
        "unresolved_fields": 1
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction"
      ],
      "coverage_notes": [
        {
          "count": 61,
          "coverage": "explicitly-in-scope",
          "notes": "CASP17 reports 61 multimer targets re-released with stoichiometry as of 2026-07-21; this is a rolling target-release count, not an evaluated final set.",
          "reporting_status": "reported",
          "tag": "protein-protein-binding"
        },
        {
          "count": null,
          "coverage": "explicitly-in-scope",
          "notes": "Immune complexes are a dedicated CASP17 category, but the official in-numbers page does not publish a separate immune-complex count.",
          "reporting_status": "not_reported",
          "tag": "antibody-antigen"
        },
        {
          "count": null,
          "coverage": "explicitly-in-scope",
          "notes": "Organic ligand-protein complexes are a dedicated CASP17 category, but no current ligand-only target count is published on the CASP17 in-numbers page.",
          "reporting_status": "not_reported",
          "tag": "protein-ligand-binding"
        }
      ],
      "domains": [
        "protein-structure",
        "protein-protein-binding",
        "protein-ligand-binding",
        "antibody-antigen"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "casp-evidence-origin",
          "locator": {
            "note": "Reports the first large-scale experiment in 1994, 33 targets, 35 groups, and independent assessment teams; no exact start day is given.",
            "type": "section",
            "value": "CASP1 description"
          },
          "source_id": "casp1-history-resource",
          "source_type": "resource",
          "supports": [
            "/name",
            "/aliases",
            "/organizations",
            "/release_date",
            "/kind",
            "/summary"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "casp-evidence-casp16-protocol",
          "locator": {
            "note": "Completed-round scope, blind protocol, categories, independent assessors, access, and first-target date.",
            "type": "section",
            "value": "Description; Modeling categories; Timetable; Targets; Assessment; Results"
          },
          "source_id": "casp-official",
          "source_type": "work",
          "supports": [
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/access",
            "/access/level",
            "/access/license",
            "/resources",
            "/implementations",
            "/versions/0/release_date",
            "/versions/0/formal_tracks"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "casp-evidence-casp16-counts",
          "locator": {
            "note": "Reports 156 tertiary, 108 multimer, 8 incidental ligand, 233 pharma pose, 140 pharma affinity, and 110 stage-2 affinity target releases, plus submission counts.",
            "type": "table",
            "value": "CASP16 in numbers"
          },
          "source_id": "casp16-numbers-resource",
          "source_type": "resource",
          "supports": [
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "casp-evidence-casp17-protocol",
          "locator": {
            "note": "Defines the active CASP17 round, dedicated immune and ligand categories, difficult structures/complexes, release schedule, and independent assessors.",
            "type": "section",
            "value": "Description; Modeling categories; Timetable; Targets; Assessment; Results"
          },
          "source_id": "casp17-official",
          "source_type": "work",
          "supports": [
            "/latest_version",
            "/domains",
            "/coverage_notes",
            "/versions/1/release_date",
            "/versions/1/as_of",
            "/versions/1/formal_tracks",
            "/scientific_task_classification/entries/0",
            "/scientific_task_classification/entries/2",
            "/scientific_task_classification/entries/3",
            "/scientific_task_classification/entries/4"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "casp-evidence-casp17-counts",
          "locator": {
            "note": "Rolling counts: 75 protein targets without stoichiometry, 61 multimer re-releases with stoichiometry, 45 RNA targets, and 8 hybrids.",
            "type": "table",
            "value": "CASP17 in numbers, accessed 2026-07-21"
          },
          "source_id": "casp17-numbers-resource",
          "source_type": "resource",
          "supports": [
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes/0",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets",
            "/scientific_task_classification/entries/1"
          ]
        }
      ],
      "field_status": [
        {
          "confidence": "medium",
          "evidence_ids": [
            "casp-evidence-origin"
          ],
          "path": "/release_date",
          "reason": "The official CASP1 history reports only the year 1994; 1994-01-01 is a machine-readable year-normalized date and not an asserted day of first target release.",
          "status": "provisional"
        }
      ],
      "id": "casp",
      "implementations": [
        {
          "commit": null,
          "framework": "Prediction Center result and assessment system",
          "notes": "Official score tables and rankings are public, but no single commit-pinned cross-category harness is published.",
          "status": "official",
          "url": "https://predictioncenter.org/casp16/index.cgi"
        }
      ],
      "kind": "competition",
      "latest_version": "CASP17",
      "modalities": [
        "protein-sequence",
        "dna-rna-sequence",
        "structure-3d"
      ],
      "name": "CASP",
      "organizations": [
        "Prediction Center",
        "CASP organizing committee"
      ],
      "parent_id": null,
      "release_date": "1994-01-01",
      "resources": [
        {
          "access_notes": "Official series portal and historical round index.",
          "id": "casp-series-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY 4.0 (website)",
          "pin": null,
          "type": "website",
          "url": "https://predictioncenter.org/"
        },
        {
          "access_notes": "Official CASP1 historical description.",
          "id": "casp1-history-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY 4.0 (website)",
          "pin": {
            "kind": "snapshot",
            "url": "https://predictioncenter.org/decoysets2019/description.cgi?casp=CASP1",
            "value": "CASP1-1994"
          },
          "type": "documentation",
          "url": "https://predictioncenter.org/decoysets2019/description.cgi?casp=CASP1"
        },
        {
          "access_notes": "Completed CASP16 competition portal.",
          "id": "casp16-home-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY 4.0 (website)",
          "pin": {
            "kind": "snapshot",
            "url": "https://predictioncenter.org/casp16/",
            "value": "CASP16"
          },
          "type": "website",
          "url": "https://predictioncenter.org/casp16/"
        },
        {
          "access_notes": "Official released-target and submission counts by category and phase.",
          "id": "casp16-numbers-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY 4.0 (website)",
          "pin": {
            "kind": "snapshot",
            "url": "https://predictioncenter.org/casp16/numbers.cgi",
            "value": "CASP16-final"
          },
          "type": "documentation",
          "url": "https://predictioncenter.org/casp16/numbers.cgi"
        },
        {
          "access_notes": "Official interactive results, rankings, and parseable-data entry point.",
          "id": "casp16-results-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY 4.0 (website)",
          "pin": {
            "kind": "snapshot",
            "url": "https://predictioncenter.org/casp16/index.cgi",
            "value": "CASP16-results"
          },
          "type": "leaderboard",
          "url": "https://predictioncenter.org/casp16/index.cgi"
        },
        {
          "access_notes": "Official round archive for targets, predictions, and results; linked artifacts may have separate terms.",
          "id": "casp16-download-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "snapshot",
            "url": "https://predictioncenter.org/download_area/CASP16/",
            "value": "CASP16-download-archive"
          },
          "type": "dataset",
          "url": "https://predictioncenter.org/download_area/CASP16/"
        },
        {
          "access_notes": "Active CASP17 portal and protocol.",
          "id": "casp17-home-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY 4.0 (website)",
          "pin": null,
          "type": "website",
          "url": "https://predictioncenter.org/casp17/"
        },
        {
          "access_notes": "Rolling official counts; snapshot date is required for interpretation.",
          "id": "casp17-numbers-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY 4.0 (website)",
          "pin": {
            "kind": "snapshot",
            "url": "https://predictioncenter.org/casp17/numbers.cgi",
            "value": "2026-07-21"
          },
          "type": "documentation",
          "url": "https://predictioncenter.org/casp17/numbers.cgi"
        },
        {
          "access_notes": "Living official target table with release and deadline metadata.",
          "id": "casp17-target-list-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY 4.0 (website metadata)",
          "pin": {
            "kind": "snapshot",
            "url": "https://predictioncenter.org/casp17/targetlist.cgi",
            "value": "2026-07-21"
          },
          "type": "dataset",
          "url": "https://predictioncenter.org/casp17/targetlist.cgi"
        }
      ],
      "scientific_task_classification": {
        "as_of": "2026-07-21",
        "benchmark_version": "CASP17",
        "entries": [
          {
            "confidence": "high",
            "count": null,
            "count_basis": "CASP17 assessed monomer evaluation units.",
            "count_ref": null,
            "count_unit": "targets",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "casp-evidence-casp17-protocol"
            ],
            "mapping_method": "official-track",
            "notes": "The final assessed monomer count is not yet reported.",
            "reporting_status": "not_reported",
            "task_type_id": "protein-monomer-structure-prediction"
          },
          {
            "confidence": "high",
            "count": 61,
            "count_basis": "CASP17 multimer target re-releases with stoichiometry as of 2026-07-21.",
            "count_ref": "/coverage_notes/0/count",
            "count_unit": "targets",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "casp-evidence-casp17-counts"
            ],
            "mapping_method": "official-track",
            "notes": "Rolling release count; not a final assessed-set total.",
            "reporting_status": "reported",
            "task_type_id": "protein-complex-structure-prediction"
          },
          {
            "confidence": "high",
            "count": null,
            "count_basis": "CASP17 model-selection and accuracy-estimation targets.",
            "count_ref": null,
            "count_unit": "targets",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "casp-evidence-casp17-protocol"
            ],
            "mapping_method": "official-track",
            "notes": "No final current-round count is asserted.",
            "reporting_status": "not_reported",
            "task_type_id": "protein-model-quality-assessment"
          },
          {
            "confidence": "high",
            "count": null,
            "count_basis": "CASP17 protein-ligand targets.",
            "count_ref": null,
            "count_unit": "targets",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "casp-evidence-casp17-protocol"
            ],
            "mapping_method": "official-track",
            "notes": "Pose prediction is a formal ligand-category objective.",
            "reporting_status": "not_reported",
            "task_type_id": "protein-ligand-pose-prediction"
          },
          {
            "confidence": "high",
            "count": null,
            "count_basis": "CASP17 protein-ligand targets.",
            "count_ref": null,
            "count_unit": "targets",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "casp-evidence-casp17-protocol"
            ],
            "mapping_method": "official-track",
            "notes": "Affinity or rank prediction is a formal ligand-category objective.",
            "reporting_status": "not_reported",
            "task_type_id": "protein-ligand-binding-affinity"
          }
        ],
        "notes": "CASP17 is active; mappings follow formal categories while final assessed target counts remain unavailable.",
        "status": "partial"
      },
      "summary": "Biennial blind community experiments that assess macromolecular structure, complex, ligand, and model-accuracy prediction against experimental structures withheld during prediction.",
      "task_counts": {
        "basis": "round-specific target releases; categories and repeated phases overlap",
        "reporting_status": "not_reported",
        "subsets": [
          {
            "basis": "target releases",
            "count": 75,
            "exclusive": false,
            "exhaustive": false,
            "id": "casp17-protein-no-stoichiometry",
            "label": "CASP17 protein targets released without stoichiometry",
            "notes": "Includes monomers and multimers; it must not be added to the with-stoichiometry count.",
            "reporting_status": "reported"
          },
          {
            "basis": "multimer target re-releases",
            "count": 61,
            "exclusive": false,
            "exhaustive": false,
            "id": "casp17-protein-with-stoichiometry",
            "label": "CASP17 protein targets released with stoichiometry",
            "notes": "Multimer targets only; these overlap with targets released earlier without stoichiometry.",
            "reporting_status": "reported"
          },
          {
            "basis": "target releases",
            "count": 45,
            "exclusive": false,
            "exhaustive": false,
            "id": "casp17-rna-targets",
            "label": "CASP17 RNA targets",
            "notes": "Includes RNA monomers and multimers.",
            "reporting_status": "reported"
          },
          {
            "basis": "target releases",
            "count": 8,
            "exclusive": false,
            "exhaustive": false,
            "id": "casp17-hybrid-targets",
            "label": "CASP17 protein-RNA-DNA hybrid targets",
            "notes": null,
            "reporting_status": "reported"
          }
        ],
        "total": null
      },
      "task_formats": [
        "blind structure prediction",
        "blind complex prediction",
        "blind ligand pose and affinity prediction",
        "model accuracy estimation"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Field-level audit completed against official Prediction Center round pages, number tables, target list, archive, and creator assessment papers.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2024-12-04",
          "evidence_ids": [
            "casp-evidence-casp16-protocol",
            "casp-evidence-casp16-counts"
          ],
          "formal_tracks": [
            "casp-protein-monomers",
            "casp-protein-multimers",
            "casp-protein-ligands"
          ],
          "id": "casp-suite-casp16",
          "label": "CASP16",
          "notes": "Completed 2024 round. Counts are release events by category/phase and are intentionally not summed.",
          "release_date": "2024-05-01",
          "status": "superseded",
          "task_counts": {
            "basis": "target releases across overlapping categories and phases",
            "reporting_status": "not_reported",
            "subsets": [
              {
                "basis": "target releases",
                "count": 156,
                "exclusive": false,
                "exhaustive": false,
                "id": "casp16-tertiary-releases",
                "label": "Protein tertiary-structure target releases",
                "notes": "35 Phase 0 multimer subunits, 76 Phase 1 targets, and 45 Phase 2 multimer subunits.",
                "reporting_status": "reported"
              },
              {
                "basis": "target releases",
                "count": 108,
                "exclusive": false,
                "exhaustive": false,
                "id": "casp16-multimer-releases",
                "label": "Protein multimer target releases",
                "notes": "30/43/35 Phase 0/1/2 releases; phases repeat biological targets.",
                "reporting_status": "reported"
              },
              {
                "basis": "target releases",
                "count": 8,
                "exclusive": false,
                "exhaustive": false,
                "id": "casp16-incidental-ligand-releases",
                "label": "Incidental ligand targets",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "target releases",
                "count": 233,
                "exclusive": false,
                "exhaustive": false,
                "id": "casp16-pharma-pose-releases",
                "label": "Pharmaceutical ligand pose targets",
                "notes": "The assessment paper reports 229 pose targets after evaluation filtering.",
                "reporting_status": "reported"
              },
              {
                "basis": "target releases",
                "count": 140,
                "exclusive": false,
                "exhaustive": false,
                "id": "casp16-pharma-affinity-releases",
                "label": "Pharmaceutical ligand affinity targets",
                "notes": "110 were re-released for stage-2 affinity prediction.",
                "reporting_status": "reported"
              }
            ],
            "total": null
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "casp-evidence-casp17-protocol",
            "casp-evidence-casp17-counts"
          ],
          "formal_tracks": [
            "casp-protein-monomers",
            "casp-protein-multimers",
            "casp-protein-ligands"
          ],
          "id": "casp-suite-casp17",
          "label": "CASP17",
          "notes": "Active 2026 round. Target release continues through 2026-07-31 and evaluation is scheduled for August-October; these counts are a dated rolling snapshot, not final evaluation counts.",
          "release_date": "2026-04-27",
          "status": "rolling",
          "task_counts": {
            "basis": "round-specific target releases; categories and repeated phases overlap",
            "reporting_status": "not_reported",
            "subsets": [
              {
                "basis": "target releases",
                "count": 75,
                "exclusive": false,
                "exhaustive": false,
                "id": "casp17-protein-no-stoichiometry",
                "label": "CASP17 protein targets released without stoichiometry",
                "notes": "Includes monomers and multimers; it must not be added to the with-stoichiometry count.",
                "reporting_status": "reported"
              },
              {
                "basis": "multimer target re-releases",
                "count": 61,
                "exclusive": false,
                "exhaustive": false,
                "id": "casp17-protein-with-stoichiometry",
                "label": "CASP17 protein targets released with stoichiometry",
                "notes": "Multimer targets only; these overlap with targets released earlier without stoichiometry.",
                "reporting_status": "reported"
              },
              {
                "basis": "target releases",
                "count": 45,
                "exclusive": false,
                "exhaustive": false,
                "id": "casp17-rna-targets",
                "label": "CASP17 RNA targets",
                "notes": "Includes RNA monomers and multimers.",
                "reporting_status": "reported"
              },
              {
                "basis": "target releases",
                "count": 8,
                "exclusive": false,
                "exhaustive": false,
                "id": "casp17-hybrid-targets",
                "label": "CASP17 protein-RNA-DNA hybrid targets",
                "notes": null,
                "reporting_status": "reported"
              }
            ],
            "total": null
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "Predictions and numerical assessment are scheduled for public release before the December 2026 meeting.",
        "biosafety_notes": "Targets include immune and viral-host systems; this registry stores metadata only and mirrors no sequences or structures.",
        "grader": "Dedicated independent immune-complex assessor; final metric set may be extended during assessment.",
        "level": "partially-open",
        "license": "CC BY 4.0 for Prediction Center website content; linked structures and predictions may have separate terms",
        "tasks": "Target sequences and deadlines are public; experimental complex structures remain blind until target closure."
      },
      "aliases": [
        "CASP17 antibody-antigen category"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Missing count and final metrics are explicitly Not reported because CASP17 is ongoing.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction"
      ],
      "coverage_notes": [
        {
          "count": null,
          "coverage": "explicitly-in-scope",
          "notes": "The category is official, but no category-only count is published on the rolling CASP17 in-numbers page.",
          "reporting_status": "not_reported",
          "tag": "antibody-antigen"
        }
      ],
      "domains": [
        "protein-structure",
        "protein-protein-binding",
        "antibody-antigen"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "casp-immune-evidence-category",
          "locator": {
            "note": "Defines antibody-antigen, nanobody-antigen, and T-cell receptor scope and names a dedicated assessor.",
            "type": "section",
            "value": "Immune Complexes; Assessment; Timetable"
          },
          "source_id": "casp17-official",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/parent_id",
            "/organizations",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/grader",
            "/versions/0/as_of",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "casp-immune-evidence-target-list",
          "locator": {
            "note": "First immune-complex target release and living target metadata.",
            "type": "table",
            "value": "Target List, first H1311 release on 2026-04-28; accessed 2026-07-21"
          },
          "source_id": "casp-immune-casp17-target-list-resource",
          "source_type": "resource",
          "supports": [
            "/release_date",
            "/versions/0/release_date",
            "/access/tasks",
            "/access/artifacts",
            "/access/license",
            "/access/biosafety_notes",
            "/resources",
            "/implementations"
          ]
        }
      ],
      "field_status": [],
      "id": "casp-immune-complexes",
      "implementations": [
        {
          "commit": null,
          "framework": "CASP17 immune-complex assessment",
          "notes": "The round is active and final scoring tables/pipeline are not yet published.",
          "status": "not-available",
          "url": null
        }
      ],
      "kind": "track",
      "latest_version": "CASP17",
      "modalities": [
        "protein-sequence",
        "structure-3d"
      ],
      "name": "CASP17 Immune Complexes",
      "organizations": [
        "Prediction Center",
        "CASP organizing committee"
      ],
      "parent_id": "casp-protein-multimers",
      "release_date": "2026-04-28",
      "resources": [
        {
          "access_notes": "Official category description and assessor.",
          "id": "casp-immune-casp17-home-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY 4.0 (website)",
          "pin": {
            "kind": "snapshot",
            "url": "https://predictioncenter.org/casp17/",
            "value": "2026-07-21"
          },
          "type": "website",
          "url": "https://predictioncenter.org/casp17/"
        },
        {
          "access_notes": "Living target table; immune complexes are released as standard assembly targets.",
          "id": "casp-immune-casp17-target-list-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY 4.0 (website metadata)",
          "pin": {
            "kind": "snapshot",
            "url": "https://predictioncenter.org/casp17/targetlist.cgi",
            "value": "2026-07-21"
          },
          "type": "dataset",
          "url": "https://predictioncenter.org/casp17/targetlist.cgi"
        }
      ],
      "scientific_task_classification": {
        "as_of": "2026-07-21",
        "benchmark_version": "CASP17",
        "entries": [
          {
            "confidence": "high",
            "count": null,
            "count_basis": "CASP17 immune-complex targets.",
            "count_ref": null,
            "count_unit": "targets",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "casp-immune-evidence-category"
            ],
            "mapping_method": "official-track",
            "notes": "The official current-round page does not publish a standalone count.",
            "reporting_status": "not_reported",
            "task_type_id": "antibody-antigen-interaction"
          }
        ],
        "notes": "Dedicated immune-complex category nested under multimer prediction.",
        "status": "complete"
      },
      "summary": "Dedicated CASP17 category for blind prediction of antibody-antigen, nanobody-antigen, and T-cell receptor complex structures.",
      "task_counts": {
        "basis": "CASP17 immune-complex targets",
        "reporting_status": "not_reported",
        "subsets": [],
        "total": null
      },
      "task_formats": [
        "blind immune complex structure prediction"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Audited against the official CASP17 category description, target list, and organizer announcement.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "casp-immune-evidence-category",
            "casp-immune-evidence-target-list"
          ],
          "formal_tracks": [],
          "id": "casp-immune-casp17",
          "label": "CASP17",
          "notes": "First dedicated CASP immune-complex category; targets are released through the assembly stream and category-only count is not reported.",
          "release_date": "2026-04-28",
          "status": "rolling",
          "task_counts": {
            "basis": "immune-complex targets",
            "reporting_status": "not_reported",
            "subsets": [],
            "total": null
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "CASP16 pose and affinity result tables and assessment papers are public; some pharmaceutical source data may retain provider terms.",
        "biosafety_notes": "No ligand sets, target structures, affinities, or submitted models are mirrored.",
        "grader": "OpenStructure ligand and pocket metrics plus independent ligand-assessor analysis.",
        "level": "partially-open",
        "license": "CC BY 4.0 for Prediction Center website content; pharmaceutical datasets and linked structures may have separate terms",
        "tasks": "Active ligand/target information is released through the target list; reference poses and affinities remain blind during prediction."
      },
      "aliases": [
        "CASP ligand track",
        "CASP pharmaceutical ligand challenge"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "CASP16 released pose targets (233) and assessed pose targets (229) are distinct quantities, not a source conflict.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction",
        "regression"
      ],
      "coverage_notes": [
        {
          "count": null,
          "coverage": "explicitly-in-scope",
          "notes": "The dedicated CASP17 category is active, but the official in-numbers page does not yet publish a ligand-only target count.",
          "reporting_status": "not_reported",
          "tag": "protein-ligand-binding"
        }
      ],
      "domains": [
        "protein-structure",
        "protein-ligand-binding",
        "medchem"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "casp16-ligand-assessment"
      ],
      "evaluation_run_ids": [
        "casp16-ligand-affinity-stage1",
        "casp16-ligand-affinity-stage2",
        "casp16-ligand-pose-regular"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "casp-ligand-evidence-casp16",
          "locator": {
            "note": "Reports 229 pose targets, 140 original affinity targets, disclosure filtering to 122 Stage-1 and 103 Stage-2 analysis cases, five systems, and pose/affinity evaluation.",
            "type": "section",
            "value": "Abstract; Sections 2.2–2.5 and 3.3; Figures 8–10"
          },
          "source_id": "casp16-ligand-assessment",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/organizations",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/access/grader",
            "/resources",
            "/implementations"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "casp-ligand-evidence-casp16-protocol",
          "locator": {
            "note": "Establishes the 2024 round start and official ligand category.",
            "type": "section",
            "value": "CASP16 Timetable and Organic Ligands"
          },
          "source_id": "casp-official",
          "source_type": "work",
          "supports": [
            "/release_date",
            "/versions/0/release_date"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "casp-ligand-evidence-casp17",
          "locator": {
            "note": "Defines the active category and confirms that final evaluation is not yet available.",
            "type": "section",
            "value": "Organic Ligand-Protein Complexes; Timetable; Assessment"
          },
          "source_id": "casp17-official",
          "source_type": "work",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/versions/1/release_date",
            "/versions/1/as_of",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/license",
            "/access/biosafety_notes",
            "/scientific_task_classification/entries/0",
            "/scientific_task_classification/entries/1"
          ]
        }
      ],
      "field_status": [],
      "id": "casp-protein-ligands",
      "implementations": [
        {
          "commit": null,
          "framework": "Prediction Center/OpenStructure ligand assessment",
          "notes": "Official results expose LDDT-PLI, BiSyRMSD, pocket measures, and affinity rankings; no public source commit is asserted.",
          "status": "official",
          "url": "https://predictioncenter.org/casp16/results.cgi?tr_type=ligand&view=targets"
        }
      ],
      "kind": "track",
      "latest_version": "CASP17",
      "modalities": [
        "protein-sequence",
        "structure-3d"
      ],
      "name": "CASP Protein-Ligand Prediction",
      "organizations": [
        "Prediction Center",
        "CASP organizing committee"
      ],
      "parent_id": "casp",
      "release_date": "2024-05-01",
      "resources": [
        {
          "access_notes": "Official per-target ligand results.",
          "id": "casp-ligand-casp16-results-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY 4.0 (website)",
          "pin": {
            "kind": "snapshot",
            "url": "https://predictioncenter.org/casp16/results.cgi?tr_type=ligand&view=targets",
            "value": "CASP16-ligand-results"
          },
          "type": "leaderboard",
          "url": "https://predictioncenter.org/casp16/results.cgi?tr_type=ligand&view=targets"
        },
        {
          "access_notes": "Final pharmaceutical pose and affinity assessor paper.",
          "id": "casp-ligand-casp16-paper-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY 4.0",
          "pin": {
            "kind": "version",
            "url": "https://doi.org/10.1002/prot.70061",
            "value": "Proteins 94(1):249-266"
          },
          "type": "paper",
          "url": "https://doi.org/10.1002/prot.70061"
        },
        {
          "access_notes": "Active Organic Ligand-Protein Complexes category.",
          "id": "casp-ligand-casp17-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY 4.0 (website)",
          "pin": {
            "kind": "snapshot",
            "url": "https://predictioncenter.org/casp17/",
            "value": "2026-07-21"
          },
          "type": "website",
          "url": "https://predictioncenter.org/casp17/"
        }
      ],
      "scientific_task_classification": {
        "as_of": "2026-07-21",
        "benchmark_version": "CASP17",
        "entries": [
          {
            "confidence": "high",
            "count": null,
            "count_basis": "CASP17 protein-ligand targets.",
            "count_ref": null,
            "count_unit": "targets",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "casp-ligand-evidence-casp17"
            ],
            "mapping_method": "official-track",
            "notes": "Pose and pocket prediction are explicit objectives.",
            "reporting_status": "not_reported",
            "task_type_id": "protein-ligand-pose-prediction"
          },
          {
            "confidence": "high",
            "count": null,
            "count_basis": "CASP17 protein-ligand targets.",
            "count_ref": null,
            "count_unit": "targets",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "casp-ligand-evidence-casp17"
            ],
            "mapping_method": "official-track",
            "notes": "Affinity or ranking is explicit, without a current standalone count.",
            "reporting_status": "not_reported",
            "task_type_id": "protein-ligand-binding-affinity"
          }
        ],
        "notes": "Dedicated protein-ligand track; current target counts are not final.",
        "status": "complete"
      },
      "summary": "Formal CASP track for blind prediction of protein-ligand binding poses, binding affinity or rank, binding pockets, and pose confidence.",
      "task_counts": {
        "basis": "CASP17 evaluated protein-ligand targets",
        "reporting_status": "not_reported",
        "subsets": [],
        "total": null
      },
      "task_formats": [
        "blind protein-ligand pose prediction",
        "blind binding affinity prediction"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Audited against official results, creator-assessor paper, and active CASP17 protocol.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2025-10-04",
          "evidence_ids": [
            "casp-ligand-evidence-casp16",
            "casp-ligand-evidence-casp16-protocol"
          ],
          "formal_tracks": [],
          "id": "casp-ligand-casp16",
          "label": "CASP16",
          "notes": "Released and assessed counts are stored separately; no synthetic pose-plus-affinity total is computed.",
          "release_date": "2024-05-01",
          "status": "superseded",
          "task_counts": {
            "basis": "evaluated protein-ligand cases across overlapping pose and affinity tasks",
            "reporting_status": "not_reported",
            "subsets": [
              {
                "basis": "protein-ligand complexes",
                "count": 229,
                "exclusive": false,
                "exhaustive": false,
                "id": "casp16-pharma-pose-assessed",
                "label": "Pharmaceutical pose targets assessed",
                "notes": "The in-numbers page lists 233 released pose targets; 229 were included in the final pharmaceutical assessment.",
                "reporting_status": "reported"
              },
              {
                "basis": "protein-ligand affinity cases",
                "count": 140,
                "exclusive": false,
                "exhaustive": false,
                "id": "casp16-pharma-affinity-original",
                "label": "Original pharmaceutical affinity targets",
                "notes": "Original 17 chymase plus 123 autotaxin cases; 18 disclosed affinities were removed from the ranking analysis.",
                "reporting_status": "reported"
              },
              {
                "basis": "undisclosed affinity cases",
                "count": 122,
                "exclusive": false,
                "exhaustive": false,
                "id": "casp16-affinity-stage1-analysis",
                "label": "Stage-1 affinity cases retained for ranking analysis",
                "notes": "14 chymase plus 108 autotaxin cases after removing 18 affinities already disclosed in patents.",
                "reporting_status": "reported"
              },
              {
                "basis": "target releases with experimental structures",
                "count": 110,
                "exclusive": false,
                "exhaustive": false,
                "id": "casp16-affinity-stage2-releases",
                "label": "Stage-2 affinity target releases",
                "notes": "Official in-numbers release count: 17 chymase plus 93 autotaxin.",
                "reporting_status": "reported"
              },
              {
                "basis": "undisclosed affinity cases with experimental structures",
                "count": 103,
                "exclusive": false,
                "exhaustive": false,
                "id": "casp16-affinity-stage2-analysis",
                "label": "Stage-2 affinity cases retained for ranking analysis",
                "notes": "14 chymase plus 89 autotaxin cases after disclosure filtering.",
                "reporting_status": "reported"
              },
              {
                "basis": "protein systems",
                "count": 5,
                "exclusive": false,
                "exhaustive": false,
                "id": "casp16-ligand-protein-systems",
                "label": "Protein systems in the pharmaceutical assessment",
                "notes": null,
                "reporting_status": "reported"
              }
            ],
            "total": null
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "casp-ligand-evidence-casp17"
          ],
          "formal_tracks": [],
          "id": "casp-ligand-casp17",
          "label": "CASP17",
          "notes": "Active round; final ligand target and assessed-case counts are not yet published.",
          "release_date": "2026-04-27",
          "status": "rolling",
          "task_counts": {
            "basis": "evaluated protein-ligand targets",
            "reporting_status": "not_reported",
            "subsets": [],
            "total": null
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "CASP16 predictions, per-model score tables, rankings, and assessor paper are public.",
        "biosafety_notes": "No active target sequences, structures, or predictions are mirrored.",
        "grader": "Official structure-comparison scores plus independent assessor review.",
        "level": "partially-open",
        "license": "CC BY 4.0 for Prediction Center website content; linked structures and predictions may have separate terms",
        "tasks": "Active target sequences are public while experimental coordinates are withheld; completed-round targets are archived."
      },
      "aliases": [
        "CASP single proteins and domains",
        "CASP protein domains"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "CASP16 assessed-unit counts are final; CASP17 remains rolling.",
        "status": "audited-with-caveats",
        "unresolved_fields": 1
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction"
      ],
      "coverage_notes": [
        {
          "count": null,
          "coverage": "explicitly-in-scope",
          "notes": "CASP17 is active and its final assessed monomer evaluation-unit count is not yet available.",
          "reporting_status": "not_reported",
          "tag": "protein-structure"
        }
      ],
      "domains": [
        "protein-structure"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "casp16-monomer-assessment"
      ],
      "evaluation_run_ids": [
        "casp16-monomer-regular-official"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "casp-monomer-evidence-origin",
          "locator": {
            "note": "The 1994 experiment included comparative modeling, fold recognition, and ab initio folding of protein targets.",
            "type": "section",
            "value": "CASP1 description"
          },
          "source_id": "casp-monomer-casp1-history-resource",
          "source_type": "resource",
          "supports": [
            "/name",
            "/aliases",
            "/organizations",
            "/release_date",
            "/kind",
            "/parent_id"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "casp-monomer-evidence-casp16",
          "locator": {
            "note": "Reports 54 assessed evaluation units, including 42 complete protein monomers, and the GDT_TS-based assessment.",
            "type": "section",
            "value": "Abstract; Results and Discussion; Figure 1"
          },
          "source_id": "casp16-monomer-assessment",
          "source_type": "work",
          "supports": [
            "/summary",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/access/grader",
            "/resources",
            "/implementations"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "casp-monomer-evidence-casp17",
          "locator": {
            "note": "Active track scope and the absence of final 2026 results.",
            "type": "section",
            "value": "Difficult Protein Structures and Complexes; Timetable; Assessment"
          },
          "source_id": "casp17-official",
          "source_type": "work",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/versions/1/release_date",
            "/versions/1/as_of",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/license",
            "/access/biosafety_notes",
            "/scientific_task_classification/entries/0"
          ]
        }
      ],
      "field_status": [
        {
          "confidence": "medium",
          "evidence_ids": [
            "casp-monomer-evidence-origin"
          ],
          "path": "/release_date",
          "reason": "The official history establishes monomeric protein structure prediction in CASP1 in 1994 but does not report an exact first-target day; the date is year-normalized.",
          "status": "provisional"
        }
      ],
      "id": "casp-protein-monomers",
      "implementations": [
        {
          "commit": null,
          "framework": "Prediction Center monomer scoring and ranking system",
          "notes": "Official parseable tables expose GDT_TS and related measures; no public source commit is asserted.",
          "status": "official",
          "url": "https://predictioncenter.org/casp16/index.cgi"
        }
      ],
      "kind": "track",
      "latest_version": "CASP17",
      "modalities": [
        "protein-sequence",
        "structure-3d"
      ],
      "name": "CASP Protein Monomers",
      "organizations": [
        "Prediction Center",
        "CASP organizing committee"
      ],
      "parent_id": "casp",
      "release_date": "1994-01-01",
      "resources": [
        {
          "access_notes": "Official CASP1 historical description.",
          "id": "casp-monomer-casp1-history-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY 4.0 (website)",
          "pin": {
            "kind": "snapshot",
            "url": "https://predictioncenter.org/decoysets2019/description.cgi?casp=CASP1",
            "value": "CASP1-1994"
          },
          "type": "documentation",
          "url": "https://predictioncenter.org/decoysets2019/description.cgi?casp=CASP1"
        },
        {
          "access_notes": "Completed CASP16 protocol and timetable.",
          "id": "casp-monomer-casp16-home-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY 4.0 (website)",
          "pin": {
            "kind": "snapshot",
            "url": "https://predictioncenter.org/casp16/",
            "value": "CASP16"
          },
          "type": "website",
          "url": "https://predictioncenter.org/casp16/"
        },
        {
          "access_notes": "Official per-evaluation-unit monomer results.",
          "id": "casp-monomer-casp16-results-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY 4.0 (website)",
          "pin": {
            "kind": "snapshot",
            "url": "https://predictioncenter.org/casp16/results.cgi?groups_id=&model=all&tr_type=all",
            "value": "CASP16-protein-monomers"
          },
          "type": "leaderboard",
          "url": "https://predictioncenter.org/casp16/results.cgi?groups_id=&model=all&tr_type=all"
        },
        {
          "access_notes": "Final creator-assessor paper.",
          "id": "casp-monomer-casp16-paper-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY 4.0",
          "pin": {
            "kind": "version",
            "url": "https://doi.org/10.1002/prot.70031",
            "value": "Proteins 94(1):86-105"
          },
          "type": "paper",
          "url": "https://doi.org/10.1002/prot.70031"
        },
        {
          "access_notes": "Active difficult protein structures and complexes category.",
          "id": "casp-monomer-casp17-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY 4.0 (website)",
          "pin": {
            "kind": "snapshot",
            "url": "https://predictioncenter.org/casp17/",
            "value": "2026-07-21"
          },
          "type": "website",
          "url": "https://predictioncenter.org/casp17/"
        }
      ],
      "scientific_task_classification": {
        "as_of": "2026-07-21",
        "benchmark_version": "CASP17",
        "entries": [
          {
            "confidence": "high",
            "count": null,
            "count_basis": "CASP17 assessed monomer evaluation units.",
            "count_ref": null,
            "count_unit": "targets",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "casp-monomer-evidence-casp17"
            ],
            "mapping_method": "official-track",
            "notes": "Final assessed count is not reported while CASP17 is active.",
            "reporting_status": "not_reported",
            "task_type_id": "protein-monomer-structure-prediction"
          }
        ],
        "notes": "Single-purpose formal track; the current round is rolling.",
        "status": "complete"
      },
      "summary": "Formal CASP track assessing blind predictions of single-protein structures and post hoc protein evaluation units against withheld experimental coordinates.",
      "task_counts": {
        "basis": "CASP17 assessed monomer evaluation units",
        "reporting_status": "not_reported",
        "subsets": [],
        "total": null
      },
      "task_formats": [
        "blind monomer structure prediction"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Audited against official round pages, results, and the final CASP16 assessor paper.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2025-08-17",
          "evidence_ids": [
            "casp-monomer-evidence-casp16"
          ],
          "formal_tracks": [],
          "id": "casp-monomer-casp16",
          "label": "CASP16",
          "notes": "Assessment units differ from the 156 tertiary target-release events reported by the round-wide in-numbers page.",
          "release_date": "2024-05-01",
          "status": "superseded",
          "task_counts": {
            "basis": "assessed protein monomer evaluation units",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "evaluation units",
                "count": 42,
                "exclusive": true,
                "exhaustive": false,
                "id": "casp16-complete-protein-monomers",
                "label": "Complete protein monomers",
                "notes": "The remaining evaluation units are subdivisions of multi-domain proteins.",
                "reporting_status": "reported"
              }
            ],
            "total": 54
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "casp-monomer-evidence-casp17"
          ],
          "formal_tracks": [],
          "id": "casp-monomer-casp17",
          "label": "CASP17",
          "notes": "Active round; final monomer evaluation units and results are not yet defined.",
          "release_date": "2026-04-27",
          "status": "rolling",
          "task_counts": {
            "basis": "assessed protein monomer evaluation units",
            "reporting_status": "not_reported",
            "subsets": [],
            "total": null
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "Completed-round predictions, OpenStructure-derived scores, rankings, and assessor papers are public.",
        "biosafety_notes": "No target sequences, immune-complex structures, or submitted models are mirrored.",
        "grader": "OpenStructure/CASP and CAPRI measures plus independent assessor review.",
        "level": "partially-open",
        "license": "CC BY 4.0 for Prediction Center website content; linked structures and predictions may have separate terms",
        "tasks": "Active sequences and phase metadata are public; experimental structures remain blind until target closure."
      },
      "aliases": [
        "CASP protein complexes",
        "CASP assembly prediction",
        "CASP oligomer prediction"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Phase-specific release counts and unique assessed targets are stored as different bases.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction"
      ],
      "coverage_notes": [
        {
          "count": 61,
          "coverage": "explicitly-in-scope",
          "notes": "Rolling CASP17 with-stoichiometry releases; this is not the final assessed target count.",
          "reporting_status": "reported",
          "tag": "protein-protein-binding"
        },
        {
          "count": null,
          "coverage": "explicitly-in-scope",
          "notes": "Immune complexes are separated into a dedicated CASP17 child track; the in-numbers page does not publish its count.",
          "reporting_status": "not_reported",
          "tag": "antibody-antigen"
        }
      ],
      "domains": [
        "protein-structure",
        "protein-protein-binding",
        "antibody-antigen"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "casp16-multimer-assessment"
      ],
      "evaluation_run_ids": [
        "casp16-multimer-phase1-regular"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "casp-multimer-evidence-casp16",
          "locator": {
            "note": "Reports 40 Phase-1 targets, 30 reused in Phase 0, 35 in Phase 2, eight antibody/nanobody-antigen targets, and DockQ/TM/lDDT/interface metrics.",
            "type": "section",
            "value": "Overview of CASP16 oligomer targets; Performance Evaluation and Ranking"
          },
          "source_id": "casp16-multimer-assessment",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/organizations",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/access/grader",
            "/resources",
            "/implementations",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "casp-multimer-evidence-casp16-protocol",
          "locator": {
            "note": "Establishes the 2024 round start and the completed multimer category protocol.",
            "type": "section",
            "value": "CASP16 Timetable and Protein Complexes"
          },
          "source_id": "casp-official",
          "source_type": "work",
          "supports": [
            "/release_date",
            "/versions/0/release_date"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "casp-multimer-evidence-casp17",
          "locator": {
            "note": "Reports 61 multimer targets released with stoichiometry.",
            "type": "table",
            "value": "CASP17 in numbers, accessed 2026-07-21"
          },
          "source_id": "casp-multimer-casp17-numbers-resource",
          "source_type": "resource",
          "supports": [
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes/0",
            "/versions/1/as_of",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "casp-multimer-evidence-casp17-protocol",
          "locator": {
            "note": "Defines the active CASP17 complex categories, public target protocol, and scheduled result release.",
            "type": "section",
            "value": "Difficult Protein Structures and Complexes; Immune Complexes; Timetable; Results"
          },
          "source_id": "casp17-official",
          "source_type": "work",
          "supports": [
            "/latest_version",
            "/versions/1/formal_tracks",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/license",
            "/access/biosafety_notes",
            "/scientific_task_classification/entries/1"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "casp-multimer-evidence-casp17-date",
          "locator": {
            "note": "Establishes the first release date used for the CASP17 multimer snapshot.",
            "type": "table",
            "value": "Target List, first protein-complex target H1311 released 2026-04-28"
          },
          "source_id": "casp-multimer-casp17-target-list-resource",
          "source_type": "resource",
          "supports": [
            "/versions/1/release_date"
          ]
        }
      ],
      "field_status": [],
      "id": "casp-protein-multimers",
      "implementations": [
        {
          "commit": null,
          "framework": "Prediction Center/OpenStructure multimer assessment",
          "notes": "Official results expose overall and interface measures; no public source commit is asserted.",
          "status": "official",
          "url": "https://predictioncenter.org/casp16/index.cgi"
        }
      ],
      "kind": "track",
      "latest_version": "CASP17",
      "modalities": [
        "protein-sequence",
        "structure-3d"
      ],
      "name": "CASP Protein Multimers",
      "organizations": [
        "Prediction Center",
        "CASP organizing committee",
        "CAPRI"
      ],
      "parent_id": "casp",
      "release_date": "2024-05-01",
      "resources": [
        {
          "access_notes": "Completed CASP16 protocol and timetable.",
          "id": "casp-multimer-casp16-home-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY 4.0 (website)",
          "pin": {
            "kind": "snapshot",
            "url": "https://predictioncenter.org/casp16/",
            "value": "CASP16"
          },
          "type": "website",
          "url": "https://predictioncenter.org/casp16/"
        },
        {
          "access_notes": "Official per-target protein multimer results.",
          "id": "casp-multimer-casp16-results-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY 4.0 (website)",
          "pin": {
            "kind": "snapshot",
            "url": "https://predictioncenter.org/casp16/results.cgi?view=targets&tr_type=multimer",
            "value": "CASP16-protein-multimers"
          },
          "type": "leaderboard",
          "url": "https://predictioncenter.org/casp16/results.cgi?view=targets&tr_type=multimer"
        },
        {
          "access_notes": "Final creator-assessor paper.",
          "id": "casp-multimer-casp16-paper-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY 4.0",
          "pin": {
            "kind": "version",
            "url": "https://doi.org/10.1002/prot.70068",
            "value": "Proteins 94(1):106-130"
          },
          "type": "paper",
          "url": "https://doi.org/10.1002/prot.70068"
        },
        {
          "access_notes": "Rolling target releases and submission counts.",
          "id": "casp-multimer-casp17-numbers-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY 4.0 (website)",
          "pin": {
            "kind": "snapshot",
            "url": "https://predictioncenter.org/casp17/numbers.cgi",
            "value": "2026-07-21"
          },
          "type": "documentation",
          "url": "https://predictioncenter.org/casp17/numbers.cgi"
        },
        {
          "access_notes": "Living official target table with release dates.",
          "id": "casp-multimer-casp17-target-list-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY 4.0 (website metadata)",
          "pin": {
            "kind": "snapshot",
            "url": "https://predictioncenter.org/casp17/targetlist.cgi",
            "value": "2026-07-21"
          },
          "type": "dataset",
          "url": "https://predictioncenter.org/casp17/targetlist.cgi"
        }
      ],
      "scientific_task_classification": {
        "as_of": "2026-07-21",
        "benchmark_version": "CASP17",
        "entries": [
          {
            "confidence": "high",
            "count": 61,
            "count_basis": "CASP17 multimer target re-releases with stoichiometry as of 2026-07-21.",
            "count_ref": "/task_counts/total",
            "count_unit": "targets",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "casp-multimer-evidence-casp17"
            ],
            "mapping_method": "official-track",
            "notes": "Rolling release count; not a final assessed-set total.",
            "reporting_status": "reported",
            "task_type_id": "protein-complex-structure-prediction"
          },
          {
            "confidence": "high",
            "count": null,
            "count_basis": "CASP17 multimer model-selection evaluation units.",
            "count_ref": null,
            "count_unit": "targets",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "casp-multimer-evidence-casp17-protocol"
            ],
            "mapping_method": "official-track",
            "notes": "Model-selection is in scope but not separately counted.",
            "reporting_status": "not_reported",
            "task_type_id": "protein-model-quality-assessment"
          }
        ],
        "notes": "Formal complex-prediction track, including interface and model-selection phases.",
        "status": "complete"
      },
      "summary": "Formal CASP track assessing blind protein-complex predictions, including overall folds, interfaces, stoichiometry-free phases, and model-selection phases.",
      "task_counts": {
        "basis": "CASP17 multimer target re-releases with stoichiometry as of 2026-07-21",
        "reporting_status": "reported",
        "subsets": [],
        "total": 61
      },
      "task_formats": [
        "blind protein complex prediction",
        "stoichiometry prediction",
        "complex model selection"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Audited against official round counts/results and the final CASP16 complex assessor paper.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2025-10-31",
          "evidence_ids": [
            "casp-multimer-evidence-casp16",
            "casp-multimer-evidence-casp16-protocol"
          ],
          "formal_tracks": [],
          "id": "casp-multimer-casp16",
          "label": "CASP16",
          "notes": "The round in-numbers page reports 108 phase-specific multimer target releases (30/43/35); the assessor paper reports 40 unique Phase-1 targets.",
          "release_date": "2024-05-01",
          "status": "superseded",
          "task_counts": {
            "basis": "unique protein complex targets in the Phase-1 main assessment",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "targets reused from Phase 1",
                "count": 30,
                "exclusive": false,
                "exhaustive": false,
                "id": "casp16-multimer-phase0",
                "label": "Phase 0 without stoichiometry",
                "notes": "Repeated targets; do not add to the 40 unique Phase-1 targets.",
                "reporting_status": "reported"
              },
              {
                "basis": "targets reused from Phase 1",
                "count": 35,
                "exclusive": false,
                "exhaustive": false,
                "id": "casp16-multimer-phase2",
                "label": "Phase 2 MassiveFold model-selection targets",
                "notes": "Repeated targets; do not add to the 40 unique Phase-1 targets.",
                "reporting_status": "reported"
              },
              {
                "basis": "Phase-1 targets",
                "count": 8,
                "exclusive": false,
                "exhaustive": false,
                "id": "casp16-antibody-antigen-targets",
                "label": "Antibody or nanobody-antigen complexes",
                "notes": "Three nanobody-antigen plus five antibody-antigen complexes.",
                "reporting_status": "reported"
              }
            ],
            "total": 40
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "casp-multimer-evidence-casp17",
            "casp-multimer-evidence-casp17-protocol",
            "casp-multimer-evidence-casp17-date"
          ],
          "formal_tracks": [
            "casp-immune-complexes"
          ],
          "id": "casp-multimer-casp17",
          "label": "CASP17",
          "notes": "Active round; 61 is a dated release count, not a final assessed-set size.",
          "release_date": "2026-04-28",
          "status": "rolling",
          "task_counts": {
            "basis": "multimer target re-releases with stoichiometry",
            "reporting_status": "reported",
            "subsets": [],
            "total": 61
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The versioned Zenodo record provides a 12,001,320,960-byte input-data archive and a 100-row TSV; a commit-pinned Hugging Face mirror exposes the task files individually.",
        "biosafety_notes": "The registry mirrors no task data or answers. The creators use synthetic/augmented data and scrubbed metadata to prevent lookup shortcuts; source-resource terms still apply.",
        "grader": "The public leaderboard scores whitespace-stripped exact string matches, but ground-truth answers and the evaluation backend are private; the public runner executes agents and extracts answers without grading them.",
        "level": "partially-open",
        "license": "CC BY 4.0 for the dataset; MIT for the runner; Apache-2.0 for the leaderboard Space; CC BY-NC 4.0 for the preprint",
        "tasks": "All 100 v1 questions and their metadata are open on Zenodo and Hugging Face."
      },
      "aliases": [
        "Computational Biology Benchmark"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Dataset, runner, leaderboard, and paper licenses are recorded separately; the runner is not mislabeled as a grader, and every result is tied to its exact agent, effort, repeat, scope, and timeout setting.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "data-analysis",
        "coding",
        "tool-use",
        "retrieval",
        "scientific-reasoning"
      ],
      "coverage_notes": [
        {
          "count": 1,
          "coverage": "observed",
          "notes": "The official v1 TSV contains one Structure-domain PDB projection task; this is not a protein-folding or binding benchmark.",
          "reporting_status": "reported",
          "tag": "protein-structure"
        },
        {
          "count": null,
          "coverage": "unknown",
          "notes": "No standalone protein-protein binding category or count is reported.",
          "reporting_status": "not_reported",
          "tag": "protein-protein-binding"
        },
        {
          "count": null,
          "coverage": "unknown",
          "notes": "No standalone protein-ligand binding category or count is reported.",
          "reporting_status": "not_reported",
          "tag": "protein-ligand-binding"
        }
      ],
      "domains": [
        "protein-structure",
        "genomics",
        "transcriptomics",
        "epigenomics",
        "single-cell",
        "spatial-omics",
        "bioinformatics",
        "clinical-translational"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "compbiobench-preprint"
      ],
      "evaluation_run_ids": [
        "compbiobench-codex-hardest",
        "compbiobench-creator-full",
        "compbiobench-haiku-full",
        "compbiobench-haiku-hardest",
        "compbiobench-nonagentic-baselines",
        "compbiobench-opus-full",
        "compbiobench-opus-hardest",
        "compbiobench-sonnet-full",
        "compbiobench-sonnet-hardest"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "compbiobench-evidence-identity",
          "locator": {
            "note": "Defines the task format, objective-answer construction, organizations, domains, capabilities, and modalities.",
            "type": "page",
            "value": "pp. 1–3, abstract, benchmark design/philosophy, and Figure 1"
          },
          "source_id": "compbiobench-preprint",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/kind",
            "/summary",
            "/capabilities",
            "/modalities"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "compbiobench-evidence-version",
          "locator": {
            "note": "Record title identifies CompBioBench v1 and links the concept DOI.",
            "type": "release",
            "value": "Zenodo immutable record 19443186, published 2026-04-06"
          },
          "source_id": "compbiobench-zenodo-resource",
          "source_type": "resource",
          "supports": [
            "/release_date",
            "/latest_version",
            "/resources/1/url",
            "/resources/1/pin"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "compbiobench-evidence-counts",
          "locator": {
            "note": "Contains 100 unique question_id rows; domain counts sum to 100, style counts sum to 100, and internet_required is 78 True / 22 False.",
            "type": "repository-path",
            "value": "compbiobench.v1.tsv (MD5 b9d72c04c018ee25798cc93ba77c1964)"
          },
          "source_id": "compbiobench-zenodo-resource",
          "source_type": "resource",
          "supports": [
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/coverage_notes/0",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "compbiobench-evidence-taxonomy",
          "locator": {
            "note": "Public questions, file paths, domain, style, skills, internet, and GPU flags.",
            "type": "repository-path",
            "value": "compbiobench.v1.tsv at commit 86ee0a22e036fef2a98f382f8fd528d5d390dde3"
          },
          "source_id": "compbiobench-hf-data-resource",
          "source_type": "resource",
          "supports": [
            "/domains",
            "/coverage_notes/1",
            "/coverage_notes/2",
            "/scientific_task_classification/entries/1"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "compbiobench-evidence-access",
          "locator": {
            "note": "Public exact-match submission UI backed by private submissions, ground-truth, and results repositories.",
            "type": "repository-path",
            "value": "app.py and README at commit 6a63e6d2cae531d9e9bf46b341606ed6b304ba3f"
          },
          "source_id": "compbiobench-leaderboard-resource",
          "source_type": "resource",
          "supports": [
            "/access/level",
            "/access/grader",
            "/access/license",
            "/resources/4/license",
            "/resources/4/pin",
            "/implementations/1"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "compbiobench-evidence-paper-license",
          "locator": {
            "note": "Preprint is CC BY-NC 4.0 and links the official data, runner, and leaderboard resources.",
            "type": "page",
            "value": "PDF page headers, first-page copyright statement, and Data availability"
          },
          "source_id": "compbiobench-preprint",
          "source_type": "work",
          "supports": [
            "/access/license",
            "/resources",
            "/resources/0/license",
            "/resources/0/pin"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "compbiobench-evidence-data-licenses",
          "locator": {
            "note": "Dataset is CC BY 4.0 and provides the immutable TSV/data-archive files.",
            "type": "release",
            "value": "Zenodo record 19443186 metadata and file manifest"
          },
          "source_id": "compbiobench-zenodo-resource",
          "source_type": "resource",
          "supports": [
            "/access/tasks",
            "/access/artifacts",
            "/access/license",
            "/resources/1/license",
            "/resources/2/license"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "compbiobench-evidence-runner-license",
          "locator": {
            "note": "MIT license and per-question Conda execution without grading.",
            "type": "repository-path",
            "value": "LICENSE, README.md, run_benchmark.py, and environment.yml at dc350ed37ccd7d7ce96347d139f06dc4bf283f26"
          },
          "source_id": "compbiobench-runner-resource",
          "source_type": "resource",
          "supports": [
            "/access/license",
            "/resources/3/license",
            "/resources/3/pin",
            "/implementations/0",
            "/scientific_task_classification/entries/0"
          ]
        }
      ],
      "field_status": [],
      "id": "compbiobench",
      "implementations": [
        {
          "commit": "dc350ed37ccd7d7ce96347d139f06dc4bf283f26",
          "framework": "CompBioBench runner",
          "notes": "Public MIT-licensed runner for Claude Code and Codex CLI with per-question Conda clones; not a hard filesystem sandbox and not a grader.",
          "status": "official",
          "url": "https://github.com/Genentech/compbiobench-runner/tree/dc350ed37ccd7d7ce96347d139f06dc4bf283f26"
        },
        {
          "commit": "6a63e6d2cae531d9e9bf46b341606ed6b304ba3f",
          "framework": "CompBioBench leaderboard",
          "notes": "Public submission interface backed by private ground-truth and result repositories.",
          "status": "official",
          "url": "https://huggingface.co/spaces/Genentech/compbiobench-leaderboard-v1/tree/6a63e6d2cae531d9e9bf46b341606ed6b304ba3f"
        }
      ],
      "kind": "agentic-eval",
      "latest_version": "v1",
      "modalities": [
        "text",
        "table",
        "dna-rna-sequence",
        "protein-sequence",
        "structure-3d",
        "raw-omics",
        "database",
        "web",
        "code"
      ],
      "name": "CompBioBench",
      "organizations": [
        "Genentech",
        "Roche"
      ],
      "parent_id": null,
      "release_date": "2026-04-06",
      "resources": [
        {
          "access_notes": "Creator preprint v1, posted 2026-04-09.",
          "id": "compbiobench-paper-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY-NC 4.0",
          "pin": {
            "kind": "snapshot",
            "url": "https://www.biorxiv.org/content/10.64898/2026.04.06.716850v1.full.pdf",
            "value": "sha256:29fcc2a5b0b2314a1810b07f338847ca3aa900d7c72c21f98e66d57f19972862"
          },
          "type": "paper",
          "url": "https://www.biorxiv.org/content/10.64898/2026.04.06.716850v1.full.pdf"
        },
        {
          "access_notes": "Immutable v1 record beneath concept DOI 10.5281/zenodo.19443185; TSV MD5 b9d72c04c018ee25798cc93ba77c1964 and data archive MD5 043dd0395898f2a71b6e81aea6a92276.",
          "id": "compbiobench-zenodo-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY 4.0",
          "pin": {
            "kind": "version",
            "url": "https://zenodo.org/records/19443186",
            "value": "v1-record-19443186"
          },
          "type": "dataset",
          "url": "https://doi.org/10.5281/zenodo.19443186"
        },
        {
          "access_notes": "Official public task and input-data mirror.",
          "id": "compbiobench-hf-data-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY 4.0",
          "pin": {
            "kind": "commit",
            "url": "https://huggingface.co/datasets/Genentech/compbiobench-data-v1/tree/86ee0a22e036fef2a98f382f8fd528d5d390dde3",
            "value": "86ee0a22e036fef2a98f382f8fd528d5d390dde3"
          },
          "type": "dataset",
          "url": "https://huggingface.co/datasets/Genentech/compbiobench-data-v1"
        },
        {
          "access_notes": "Official execution runner; it records answers, traces, time, tokens, and cost but does not contain the answer key or scoring logic.",
          "id": "compbiobench-runner-resource",
          "last_checked": "2026-07-21",
          "license": "MIT",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Genentech/compbiobench-runner/tree/dc350ed37ccd7d7ce96347d139f06dc4bf283f26",
            "value": "dc350ed37ccd7d7ce96347d139f06dc4bf283f26"
          },
          "type": "repository",
          "url": "https://github.com/Genentech/compbiobench-runner"
        },
        {
          "access_notes": "Official submission UI and exact-match scoring service; public app code submits to private answer/result repositories.",
          "id": "compbiobench-leaderboard-resource",
          "last_checked": "2026-07-21",
          "license": "Apache-2.0",
          "pin": {
            "kind": "commit",
            "url": "https://huggingface.co/spaces/Genentech/compbiobench-leaderboard-v1/tree/6a63e6d2cae531d9e9bf46b341606ed6b304ba3f",
            "value": "6a63e6d2cae531d9e9bf46b341606ed6b304ba3f"
          },
          "type": "leaderboard",
          "url": "https://huggingface.co/spaces/Genentech/compbiobench-leaderboard-v1"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "v1",
        "entries": [
          {
            "confidence": "high",
            "count": 100,
            "count_basis": "v1 independent computational-biology tasks",
            "count_ref": "/task_counts/total",
            "count_unit": "tasks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "compbiobench-evidence-counts",
              "compbiobench-evidence-runner-license"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "Tasks require code, tools, and external resources.",
            "reporting_status": "reported",
            "task_type_id": "end-to-end-computational-analysis"
          },
          {
            "confidence": "high",
            "count": null,
            "count_basis": "v1 tasks in official genomics, transcriptomics, epigenetics, single-cell, and spatial domains.",
            "count_ref": null,
            "count_unit": "tasks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "compbiobench-evidence-taxonomy"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "The official domains establish broad omics coverage but do not provide an exhaustive Scientific Task breakdown.",
            "reporting_status": "not_reported",
            "task_type_id": "omics-cellular-analysis"
          }
        ],
        "notes": "Official domains are not an exhaustive scientific-task taxonomy; the runner establishes the end-to-end workflow.",
        "status": "partial"
      },
      "summary": "A 100-task agent benchmark of objectively gradable computational-biology problems requiring multi-step reasoning, bespoke code, tools, and real-world external resources.",
      "task_counts": {
        "basis": "v1 independent computational-biology tasks",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "tasks in the domain partition",
            "count": 20,
            "exclusive": false,
            "exhaustive": false,
            "id": "compbiobench-domain-epigenomics",
            "label": "Domain — Epigenomics",
            "notes": "Official v1 TSV domain label; partition members are non-exhaustive here because other independent partitions are also registered.",
            "reporting_status": "reported"
          },
          {
            "basis": "tasks in the domain partition",
            "count": 20,
            "exclusive": false,
            "exhaustive": false,
            "id": "compbiobench-domain-genomics",
            "label": "Domain — Genomics",
            "notes": "Official v1 TSV domain label.",
            "reporting_status": "reported"
          },
          {
            "basis": "tasks in the domain partition",
            "count": 7,
            "exclusive": false,
            "exhaustive": false,
            "id": "compbiobench-domain-machine-learning",
            "label": "Domain — Machine Learning",
            "notes": "Official v1 TSV domain label.",
            "reporting_status": "reported"
          },
          {
            "basis": "tasks in the domain partition",
            "count": 12,
            "exclusive": false,
            "exhaustive": false,
            "id": "compbiobench-domain-population-genetics",
            "label": "Domain — Population Genetics",
            "notes": "Official v1 TSV domain label.",
            "reporting_status": "reported"
          },
          {
            "basis": "tasks in the domain partition",
            "count": 21,
            "exclusive": false,
            "exhaustive": false,
            "id": "compbiobench-domain-single-cell",
            "label": "Domain — Single-cell",
            "notes": "Official v1 TSV domain label.",
            "reporting_status": "reported"
          },
          {
            "basis": "tasks in the domain partition",
            "count": 2,
            "exclusive": false,
            "exhaustive": false,
            "id": "compbiobench-domain-spatial",
            "label": "Domain — Spatial",
            "notes": "Official v1 TSV domain label.",
            "reporting_status": "reported"
          },
          {
            "basis": "tasks in the domain partition",
            "count": 1,
            "exclusive": false,
            "exhaustive": false,
            "id": "compbiobench-domain-structure",
            "label": "Domain — Structure",
            "notes": "The sole Structure task asks which uppercase letter a supplied PDB protein resembles across projections.",
            "reporting_status": "reported"
          },
          {
            "basis": "tasks in the domain partition",
            "count": 17,
            "exclusive": false,
            "exhaustive": false,
            "id": "compbiobench-domain-transcriptomics",
            "label": "Domain — Transcriptomics",
            "notes": "Official v1 TSV domain label.",
            "reporting_status": "reported"
          },
          {
            "basis": "tasks in the question-style partition",
            "count": 27,
            "exclusive": false,
            "exhaustive": false,
            "id": "compbiobench-style-metadata-recovery",
            "label": "Question style — Metadata Recovery",
            "notes": "Official v1 TSV question_style label.",
            "reporting_status": "reported"
          },
          {
            "basis": "tasks in the question-style partition",
            "count": 17,
            "exclusive": false,
            "exhaustive": false,
            "id": "compbiobench-style-retrieval",
            "label": "Question style — Retrieval",
            "notes": "Official v1 TSV question_style label.",
            "reporting_status": "reported"
          },
          {
            "basis": "tasks in the question-style partition",
            "count": 22,
            "exclusive": false,
            "exhaustive": false,
            "id": "compbiobench-style-routine-analysis",
            "label": "Question style — Routine Analysis",
            "notes": "Official v1 TSV question_style label.",
            "reporting_status": "reported"
          },
          {
            "basis": "tasks in the question-style partition",
            "count": 26,
            "exclusive": false,
            "exhaustive": false,
            "id": "compbiobench-style-synthetic-augmented",
            "label": "Question style — Synthetic/Augmented Data",
            "notes": "Official v1 TSV question_style label.",
            "reporting_status": "reported"
          },
          {
            "basis": "tasks in the question-style partition",
            "count": 8,
            "exclusive": false,
            "exhaustive": false,
            "id": "compbiobench-style-tooling",
            "label": "Question style — Tooling",
            "notes": "Official v1 TSV question_style label.",
            "reporting_status": "reported"
          },
          {
            "basis": "tasks in the paper difficulty partition",
            "count": 17,
            "exclusive": false,
            "exhaustive": false,
            "id": "compbiobench-difficulty-level-1",
            "label": "Contributor difficulty — Level 1",
            "notes": "Contributor ratings are approximate and were not extensively calibrated.",
            "reporting_status": "reported"
          },
          {
            "basis": "tasks in the paper difficulty partition",
            "count": 26,
            "exclusive": false,
            "exhaustive": false,
            "id": "compbiobench-difficulty-level-2",
            "label": "Contributor difficulty — Level 2",
            "notes": "Contributor ratings are approximate and were not extensively calibrated.",
            "reporting_status": "reported"
          },
          {
            "basis": "tasks in the paper difficulty partition",
            "count": 40,
            "exclusive": false,
            "exhaustive": false,
            "id": "compbiobench-difficulty-level-3",
            "label": "Contributor difficulty — Level 3",
            "notes": "Contributor ratings are approximate and were not extensively calibrated.",
            "reporting_status": "reported"
          },
          {
            "basis": "tasks in the grouped hardest subset",
            "count": 17,
            "exclusive": false,
            "exhaustive": false,
            "id": "compbiobench-difficulty-levels-4-5",
            "label": "Contributor difficulty — Levels 4–5",
            "notes": "The paper groups Levels 4 and 5 for its hardest-subset analysis.",
            "reporting_status": "reported"
          },
          {
            "basis": "tasks whose v1 TSV internet_required field is True",
            "count": 78,
            "exclusive": false,
            "exhaustive": false,
            "id": "compbiobench-internet-required",
            "label": "Internet required",
            "notes": "Public v1 TSV count.",
            "reporting_status": "reported"
          },
          {
            "basis": "tasks whose v1 TSV internet_required field is False",
            "count": 22,
            "exclusive": false,
            "exhaustive": false,
            "id": "compbiobench-internet-not-required",
            "label": "Internet not required",
            "notes": "Public v1 TSV count.",
            "reporting_status": "reported"
          }
        ],
        "total": 100
      },
      "task_formats": [
        "agentic computational analysis",
        "exact single-line answer"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Field-level audit completed against bioRxiv v1, Zenodo record 19443186, the official TSV, runner commit, Hugging Face dataset, and leaderboard Space.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-04-06",
          "evidence_ids": [
            "compbiobench-evidence-version",
            "compbiobench-evidence-counts"
          ],
          "formal_tracks": [],
          "id": "compbiobench-v1",
          "label": "v1",
          "notes": "The immutable Zenodo v1 record and commit-pinned Hugging Face mirror contain the same 100-row TSV; no newer formal benchmark version was identified.",
          "release_date": "2026-04-06",
          "status": "current",
          "task_counts": {
            "basis": "v1 independent computational-biology tasks",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "tasks in the domain partition",
                "count": 20,
                "exclusive": false,
                "exhaustive": false,
                "id": "compbiobench-domain-epigenomics",
                "label": "Domain — Epigenomics",
                "notes": "Official v1 TSV domain label; partition members are non-exhaustive here because other independent partitions are also registered.",
                "reporting_status": "reported"
              },
              {
                "basis": "tasks in the domain partition",
                "count": 20,
                "exclusive": false,
                "exhaustive": false,
                "id": "compbiobench-domain-genomics",
                "label": "Domain — Genomics",
                "notes": "Official v1 TSV domain label.",
                "reporting_status": "reported"
              },
              {
                "basis": "tasks in the domain partition",
                "count": 7,
                "exclusive": false,
                "exhaustive": false,
                "id": "compbiobench-domain-machine-learning",
                "label": "Domain — Machine Learning",
                "notes": "Official v1 TSV domain label.",
                "reporting_status": "reported"
              },
              {
                "basis": "tasks in the domain partition",
                "count": 12,
                "exclusive": false,
                "exhaustive": false,
                "id": "compbiobench-domain-population-genetics",
                "label": "Domain — Population Genetics",
                "notes": "Official v1 TSV domain label.",
                "reporting_status": "reported"
              },
              {
                "basis": "tasks in the domain partition",
                "count": 21,
                "exclusive": false,
                "exhaustive": false,
                "id": "compbiobench-domain-single-cell",
                "label": "Domain — Single-cell",
                "notes": "Official v1 TSV domain label.",
                "reporting_status": "reported"
              },
              {
                "basis": "tasks in the domain partition",
                "count": 2,
                "exclusive": false,
                "exhaustive": false,
                "id": "compbiobench-domain-spatial",
                "label": "Domain — Spatial",
                "notes": "Official v1 TSV domain label.",
                "reporting_status": "reported"
              },
              {
                "basis": "tasks in the domain partition",
                "count": 1,
                "exclusive": false,
                "exhaustive": false,
                "id": "compbiobench-domain-structure",
                "label": "Domain — Structure",
                "notes": "The sole Structure task asks which uppercase letter a supplied PDB protein resembles across projections.",
                "reporting_status": "reported"
              },
              {
                "basis": "tasks in the domain partition",
                "count": 17,
                "exclusive": false,
                "exhaustive": false,
                "id": "compbiobench-domain-transcriptomics",
                "label": "Domain — Transcriptomics",
                "notes": "Official v1 TSV domain label.",
                "reporting_status": "reported"
              },
              {
                "basis": "tasks in the question-style partition",
                "count": 27,
                "exclusive": false,
                "exhaustive": false,
                "id": "compbiobench-style-metadata-recovery",
                "label": "Question style — Metadata Recovery",
                "notes": "Official v1 TSV question_style label.",
                "reporting_status": "reported"
              },
              {
                "basis": "tasks in the question-style partition",
                "count": 17,
                "exclusive": false,
                "exhaustive": false,
                "id": "compbiobench-style-retrieval",
                "label": "Question style — Retrieval",
                "notes": "Official v1 TSV question_style label.",
                "reporting_status": "reported"
              },
              {
                "basis": "tasks in the question-style partition",
                "count": 22,
                "exclusive": false,
                "exhaustive": false,
                "id": "compbiobench-style-routine-analysis",
                "label": "Question style — Routine Analysis",
                "notes": "Official v1 TSV question_style label.",
                "reporting_status": "reported"
              },
              {
                "basis": "tasks in the question-style partition",
                "count": 26,
                "exclusive": false,
                "exhaustive": false,
                "id": "compbiobench-style-synthetic-augmented",
                "label": "Question style — Synthetic/Augmented Data",
                "notes": "Official v1 TSV question_style label.",
                "reporting_status": "reported"
              },
              {
                "basis": "tasks in the question-style partition",
                "count": 8,
                "exclusive": false,
                "exhaustive": false,
                "id": "compbiobench-style-tooling",
                "label": "Question style — Tooling",
                "notes": "Official v1 TSV question_style label.",
                "reporting_status": "reported"
              },
              {
                "basis": "tasks in the paper difficulty partition",
                "count": 17,
                "exclusive": false,
                "exhaustive": false,
                "id": "compbiobench-difficulty-level-1",
                "label": "Contributor difficulty — Level 1",
                "notes": "Contributor ratings are approximate and were not extensively calibrated.",
                "reporting_status": "reported"
              },
              {
                "basis": "tasks in the paper difficulty partition",
                "count": 26,
                "exclusive": false,
                "exhaustive": false,
                "id": "compbiobench-difficulty-level-2",
                "label": "Contributor difficulty — Level 2",
                "notes": "Contributor ratings are approximate and were not extensively calibrated.",
                "reporting_status": "reported"
              },
              {
                "basis": "tasks in the paper difficulty partition",
                "count": 40,
                "exclusive": false,
                "exhaustive": false,
                "id": "compbiobench-difficulty-level-3",
                "label": "Contributor difficulty — Level 3",
                "notes": "Contributor ratings are approximate and were not extensively calibrated.",
                "reporting_status": "reported"
              },
              {
                "basis": "tasks in the grouped hardest subset",
                "count": 17,
                "exclusive": false,
                "exhaustive": false,
                "id": "compbiobench-difficulty-levels-4-5",
                "label": "Contributor difficulty — Levels 4–5",
                "notes": "The paper groups Levels 4 and 5 for its hardest-subset analysis.",
                "reporting_status": "reported"
              },
              {
                "basis": "tasks whose v1 TSV internet_required field is True",
                "count": 78,
                "exclusive": false,
                "exhaustive": false,
                "id": "compbiobench-internet-required",
                "label": "Internet required",
                "notes": "Public v1 TSV count.",
                "reporting_status": "reported"
              },
              {
                "basis": "tasks whose v1 TSV internet_required field is False",
                "count": 22,
                "exclusive": false,
                "exhaustive": false,
                "id": "compbiobench-internet-not-required",
                "label": "Internet not required",
                "notes": "Public v1 TSV count.",
                "reporting_status": "reported"
              }
            ],
            "total": 100
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "simulation code; analysis code; plot-generation code",
        "biosafety_notes": null,
        "grader": "Not established by the independently verified metadata claims.",
        "level": "fully-open",
        "license": "GNU General Public License, Version 3",
        "tasks": "See the linked official creator resources."
      },
      "aliases": [],
      "audit": {
        "audited_date": "2026-08-06",
        "notes": "Automated double-pass extraction plus deterministic official-resource pin; owner approval required.",
        "status": "audited-with-caveats",
        "unresolved_fields": 3
      },
      "benchmark_use_ids": [
        "comprehensive-benchmark-of-differential-transcript-usa-comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-1-use",
        "comprehensive-benchmark-of-differential-transcript-usa-comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-2-use",
        "comprehensive-benchmark-of-differential-transcript-usa-comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-3-use",
        "comprehensive-benchmark-of-differential-transcript-usa-comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-4-use",
        "comprehensive-benchmark-of-differential-transcript-usa-comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-5-use",
        "comprehensive-benchmark-of-differential-transcript-usa-comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-6-use"
      ],
      "capabilities": [
        "data-analysis"
      ],
      "coverage_notes": [],
      "domains": [
        "transcriptomics",
        "single-cell"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "comprehensive-benchmark-of-differential-transcript-usa"
      ],
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-08-06",
          "id": "comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-automated-metadata-1-evidence",
          "locator": {
            "document_page": 12,
            "note": null,
            "printed_page": "11",
            "source_fragment_sha256": "0d2ee9077084145d08c6a7c86062c69d3b1a97d157a91bbec9040bb0607a00c5",
            "type": "section",
            "value": "Data availability"
          },
          "source_id": "comprehensive-benchmark-of-differential-transcript-usa",
          "source_type": "work",
          "supports": [
            "/access/artifacts",
            "/access/level"
          ]
        },
        {
          "accessed_date": "2026-08-06",
          "id": "comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-automated-metadata-2-evidence",
          "locator": {
            "document_page": 13,
            "note": null,
            "printed_page": "12",
            "source_fragment_sha256": "5a72ea84ad1ef331941440cdfb9fbe76a3c54bd69f234364926c7ac32a5dc7c2",
            "type": "section",
            "value": "Data availability"
          },
          "source_id": "comprehensive-benchmark-of-differential-transcript-usa",
          "source_type": "work",
          "supports": [
            "/access/license"
          ]
        },
        {
          "accessed_date": "2026-08-06",
          "id": "comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-automated-metadata-3-evidence",
          "locator": {
            "document_page": 2,
            "note": null,
            "printed_page": "1",
            "source_fragment_sha256": "d7aea9e01a471e8110d0c22dd07f50b81c861fb924559ce44adc6fb73bfec448",
            "type": "section",
            "value": "Abstract"
          },
          "source_id": "comprehensive-benchmark-of-differential-transcript-usa",
          "source_type": "work",
          "supports": [
            "/capabilities"
          ]
        },
        {
          "accessed_date": "2026-08-06",
          "id": "comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-automated-metadata-4-evidence",
          "locator": {
            "document_page": 2,
            "note": null,
            "printed_page": "1",
            "source_fragment_sha256": "dd32e3a118f30ddf81c0e015a9114cb9a67a86b35b69b56f08aaf4de485248f4",
            "type": "section",
            "value": "Abstract"
          },
          "source_id": "comprehensive-benchmark-of-differential-transcript-usa",
          "source_type": "work",
          "supports": [
            "/domains"
          ]
        },
        {
          "accessed_date": "2026-08-06",
          "id": "comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-automated-metadata-5-evidence",
          "locator": {
            "document_page": 12,
            "note": null,
            "printed_page": "11",
            "source_fragment_sha256": "32b1082a1a92a9f6f1a45ad96b97d2b92b3bd59f031770ce0d3662ad8f73c01d",
            "type": "section",
            "value": "Conclusion"
          },
          "source_id": "comprehensive-benchmark-of-differential-transcript-usa",
          "source_type": "work",
          "supports": [
            "/kind"
          ]
        },
        {
          "accessed_date": "2026-08-06",
          "id": "comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-automated-metadata-6-evidence",
          "locator": {
            "document_page": 2,
            "note": null,
            "printed_page": "1",
            "source_fragment_sha256": "5bc8020f549ac1d3aa6c5ea744ad5f61b8cf483228db683f900e2a693f9be310",
            "type": "section",
            "value": "Abstract"
          },
          "source_id": "comprehensive-benchmark-of-differential-transcript-usa",
          "source_type": "work",
          "supports": [
            "/modalities"
          ]
        },
        {
          "accessed_date": "2026-08-06",
          "id": "comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-automated-metadata-7-evidence",
          "locator": {
            "document_page": 2,
            "note": null,
            "printed_page": "1",
            "source_fragment_sha256": "e35160733c655a0caa8e257f1592e50e98d8594e3c2bd3fca558e7350684c6c0",
            "type": "page",
            "value": "Article title"
          },
          "source_id": "comprehensive-benchmark-of-differential-transcript-usa",
          "source_type": "work",
          "supports": [
            "/name"
          ]
        },
        {
          "accessed_date": "2026-08-06",
          "id": "comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-automated-metadata-8-evidence",
          "locator": {
            "document_page": 2,
            "note": null,
            "printed_page": "1",
            "source_fragment_sha256": "6bdda4b47662a84ac85bad25907a4b8cb22e30d84b3a5d6c95b2871ce3debb7e",
            "type": "page",
            "value": "Author names, affiliation numbers, and affiliations 1–7"
          },
          "source_id": "comprehensive-benchmark-of-differential-transcript-usa",
          "source_type": "work",
          "supports": [
            "/organizations"
          ]
        },
        {
          "accessed_date": "2026-08-06",
          "id": "comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-automated-metadata-9-evidence",
          "locator": {
            "document_page": 2,
            "note": null,
            "printed_page": "1",
            "source_fragment_sha256": "7bd9181ab1e7b2916274ee2be7b1b1acb6eee3995d9ee8ac081ed525d78239d5",
            "type": "section",
            "value": "Abstract"
          },
          "source_id": "comprehensive-benchmark-of-differential-transcript-usa",
          "source_type": "work",
          "supports": [
            "/summary"
          ]
        },
        {
          "accessed_date": "2026-08-06",
          "id": "comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-automated-bibliographic-evidence",
          "locator": {
            "document_page": null,
            "note": "Resolved from the canonical paper identifier during intake.",
            "printed_page": null,
            "source_fragment_sha256": "024cc20ffcadeaa4a6ada869e4a789bce2627d4f853d78f6f06b96ceb8b44470",
            "type": "other",
            "value": "Crossref bibliographic metadata"
          },
          "source_id": "comprehensive-benchmark-of-differential-transcript-usa",
          "source_type": "work",
          "supports": [
            "/release_date"
          ]
        },
        {
          "accessed_date": "2026-08-06",
          "id": "comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-automated-count-evidence",
          "locator": {
            "document_page": 2,
            "note": null,
            "printed_page": "1",
            "source_fragment_sha256": "7bd9181ab1e7b2916274ee2be7b1b1acb6eee3995d9ee8ac081ed525d78239d5",
            "type": "section",
            "value": "Abstract"
          },
          "source_id": "comprehensive-benchmark-of-differential-transcript-usa",
          "source_type": "work",
          "supports": [
            "/task_counts/total",
            "/task_counts/basis",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis"
          ]
        },
        {
          "accessed_date": "2026-08-06",
          "id": "comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-automated-subset-coverage-evidence",
          "locator": {
            "document_page": 2,
            "note": null,
            "printed_page": "1",
            "source_fragment_sha256": "7bd9181ab1e7b2916274ee2be7b1b1acb6eee3995d9ee8ac081ed525d78239d5",
            "type": "section",
            "value": "Abstract"
          },
          "source_id": "comprehensive-benchmark-of-differential-transcript-usa",
          "source_type": "work",
          "supports": [
            "/task_counts/subsets",
            "/versions/0/task_counts/subsets"
          ]
        },
        {
          "accessed_date": "2026-08-06",
          "id": "comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-automated-resource-evidence",
          "locator": {
            "document_page": 12,
            "note": null,
            "printed_page": "11–12",
            "source_fragment_sha256": "99a372d839aadbaf21864f922cff58080ca3c54eb5c761eca0873b536d056e97",
            "type": "other",
            "value": "Data availability, document pages 12–13"
          },
          "source_id": "comprehensive-benchmark-of-differential-transcript-usa",
          "source_type": "work",
          "supports": [
            "/resources",
            "/implementations",
            "/access/level"
          ]
        },
        {
          "accessed_date": "2026-08-06",
          "id": "comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-automated-creator-evidence",
          "locator": {
            "document_page": 2,
            "note": null,
            "printed_page": "1",
            "source_fragment_sha256": "57fc09feabe9fe234c00a4beddf80bf6b2194aac87a83d27d334c1b293dae9be",
            "type": "page",
            "value": "Article DOI header"
          },
          "source_id": "comprehensive-benchmark-of-differential-transcript-usa",
          "source_type": "work",
          "supports": [
            "/resources/0"
          ]
        },
        {
          "accessed_date": "2026-08-06",
          "id": "comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-automated-version-evidence",
          "locator": {
            "document_page": 2,
            "note": null,
            "printed_page": "1",
            "source_fragment_sha256": "57fc09feabe9fe234c00a4beddf80bf6b2194aac87a83d27d334c1b293dae9be",
            "type": "page",
            "value": "Article DOI header"
          },
          "source_id": "comprehensive-benchmark-of-differential-transcript-usa",
          "source_type": "work",
          "supports": [
            "/latest_version",
            "/versions/0"
          ]
        }
      ],
      "field_status": [
        {
          "confidence": "medium",
          "evidence_ids": [
            "comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-automated-metadata-5-evidence"
          ],
          "path": "/kind",
          "reason": "The creator source and independent verifier support the Registry suite mapping, but the extractor assigned medium confidence because the source describes a benchmarking analysis and workflow rather than using the controlled word suite.",
          "status": "provisional"
        },
        {
          "confidence": "medium",
          "evidence_ids": [
            "comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-automated-metadata-1-evidence",
            "comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-automated-resource-evidence"
          ],
          "path": "/access/level",
          "reason": "The official resource and source-located access descriptions are public, but fully-open is an Atlas-controlled classification rather than a label stated verbatim by the creator source.",
          "status": "provisional"
        },
        {
          "confidence": "high",
          "evidence_ids": [
            "comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-automated-subset-coverage-evidence"
          ],
          "path": "/task_counts/subsets",
          "reason": "The double-pass review did not establish an exhaustive formal-subset inventory; the empty list must not be interpreted as evidence that the benchmark has no subsets.",
          "status": "provisional"
        }
      ],
      "id": "comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu",
      "implementations": [
        {
          "commit": "e4116544113740084c9118a1192e4de2a347edd3",
          "framework": "official repository",
          "notes": "Commit resolved deterministically during paper intake.",
          "status": "official",
          "url": "https://github.com/yollct/diffIsoUsage_benchmark"
        }
      ],
      "kind": "suite",
      "latest_version": "initial-release",
      "modalities": [
        "raw-omics"
      ],
      "name": "Comprehensive benchmark of differential transcript usage analysis for bulk and single-cell RNA sequencing",
      "organizations": [
        "Data Science in Systems Biology, Technical University of Munich",
        "Chair of Computational Systems Biology, University of Hamburg",
        "Institute for Advanced Study, Technical University of Munich",
        "National Institute of Diabetes, Digestive, and Kidney Diseases, National Institutes of Health",
        "Institute of Mathematics and Computer Science, University of Southern Denmark",
        "Munich Data Science Institute (MDSI), Technical University of Munich",
        "Department of Computer Science, Bioinformatics, Vrije Universiteit Amsterdam"
      ],
      "parent_id": null,
      "release_date": "2025-09-11",
      "resources": [
        {
          "access_notes": "Versioned creator source; full text is not mirrored.",
          "id": "comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-creator-paper-resource",
          "last_checked": "2026-08-06",
          "license": null,
          "pin": null,
          "type": "paper",
          "url": "https://doi.org/10.1093/nargab/lqaf117"
        },
        {
          "access_notes": "Official resource pinned immutably during intake.",
          "id": "comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-official-repository-resource",
          "last_checked": "2026-08-06",
          "license": "GNU General Public License, Version 3",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/yollct/diffIsoUsage_benchmark/commit/e4116544113740084c9118a1192e4de2a347edd3",
            "value": "e4116544113740084c9118a1192e4de2a347edd3"
          },
          "type": "repository",
          "url": "https://github.com/yollct/diffIsoUsage_benchmark"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "initial-release",
        "entries": [],
        "notes": "Candidate artifact-derived Scientific Task mappings lacked an official artifact-level locator and were omitted; task mapping remains pending a targeted official-artifact audit.",
        "status": "partial"
      },
      "summary": "A reusable benchmark for comparing differential transcript usage detection tools across simulated and real transcriptomics data.",
      "task_counts": {
        "basis": "The reusable benchmark spans simulated and real transcriptomics settings; no single finite primary item inventory is reported.",
        "reporting_status": "not_reported",
        "subsets": [],
        "total": null
      },
      "task_formats": [
        "unclassified"
      ],
      "verification": {
        "last_verified": "2026-08-06",
        "notes": "New family admitted only after creator source, official resource, and owner PR review.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": null,
          "evidence_ids": [
            "comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-automated-count-evidence",
            "comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-automated-version-evidence"
          ],
          "formal_tracks": [],
          "id": "comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-initial-release-version",
          "label": "initial-release",
          "notes": "Atlas snapshot label for the creator-paper release; the source does not report a formal benchmark version.",
          "release_date": "2025-09-11",
          "status": "current",
          "task_counts": {
            "basis": "The reusable benchmark spans simulated and real transcriptomics settings; no single finite primary item inventory is reported.",
            "reporting_status": "not_reported",
            "subsets": [],
            "total": null
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "Crafted datasets and source code for constructing experiments and benchmarking feature-selection methods",
        "biosafety_notes": null,
        "grader": "Not established by the independently verified metadata claims.",
        "level": "fully-open",
        "license": null,
        "tasks": "Benchmarking feature-selection methods for recovery of known crafted cell groups"
      },
      "aliases": [],
      "audit": {
        "audited_date": "2026-07-31",
        "notes": "Automated double-pass extraction plus deterministic official-resource pin; the benchmark license remains unverified and is published as null.",
        "status": "audited-with-caveats",
        "unresolved_fields": 2
      },
      "benchmark_use_ids": [
        "crafted-experiments-to-evaluate-feature-selection-meth-crafted-experiments-1-use",
        "crafted-experiments-to-evaluate-feature-selection-meth-crafted-experiments-2-use"
      ],
      "capabilities": [
        "data-analysis"
      ],
      "coverage_notes": [],
      "domains": [
        "life-science",
        "transcriptomics",
        "single-cell",
        "bioinformatics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "crafted-experiments-to-evaluate-feature-selection-meth"
      ],
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-31",
          "id": "crafted-experiments-automated-metadata-1-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": "lqaf023",
            "source_fragment_sha256": "a33f0e7cd4598770b19e8ebd49922aef263968fe4179679d886c7668e9ab52e2",
            "type": "section",
            "value": "Data availability; Code availability"
          },
          "source_id": "crafted-experiments-to-evaluate-feature-selection-meth",
          "source_type": "work",
          "supports": [
            "/access/artifacts"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "crafted-experiments-automated-metadata-2-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": "lqaf023",
            "source_fragment_sha256": "af14d8d61793a9f557d885a54cf2cdb2217108cecf08588810958e6ba2569d5f",
            "type": "section",
            "value": "Data availability"
          },
          "source_id": "crafted-experiments-to-evaluate-feature-selection-meth",
          "source_type": "work",
          "supports": [
            "/access/level"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "crafted-experiments-automated-metadata-3-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": "lqaf023",
            "source_fragment_sha256": "07d222480191ea894fd47bcd16229209a80acab8a39b72bfd4caaf4655f1401c",
            "type": "section",
            "value": "Results — Crafted experiment applications"
          },
          "source_id": "crafted-experiments-to-evaluate-feature-selection-meth",
          "source_type": "work",
          "supports": [
            "/access/tasks"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "crafted-experiments-automated-metadata-4-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": "lqaf023",
            "source_fragment_sha256": "0c5c0bffcf40f0951c69ed114fba8f00255e9ffdb47c3ac8b0cf7a82736ac151",
            "type": "section",
            "value": "Introduction"
          },
          "source_id": "crafted-experiments-to-evaluate-feature-selection-meth",
          "source_type": "work",
          "supports": [
            "/capabilities"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "crafted-experiments-automated-metadata-5-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": "lqaf023",
            "source_fragment_sha256": "3e1b3413c90f9990f1724b500d82e06b73f746d43070107b833ac0470200b1a8",
            "type": "section",
            "value": "Introduction"
          },
          "source_id": "crafted-experiments-to-evaluate-feature-selection-meth",
          "source_type": "work",
          "supports": [
            "/domains"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "crafted-experiments-automated-metadata-6-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": "lqaf023",
            "source_fragment_sha256": "06f529272f7b8982461611c8ed44a3f2c517e8a83b90a853d44e8cb41445446f",
            "type": "section",
            "value": "Results — Construction of crafted experiments"
          },
          "source_id": "crafted-experiments-to-evaluate-feature-selection-meth",
          "source_type": "work",
          "supports": [
            "/kind"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "crafted-experiments-automated-metadata-7-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": "lqaf023",
            "source_fragment_sha256": "a495e559b9cad223eb8217d0bb5dece84fde9bbfb62397a61aa41993ab8afffb",
            "type": "section",
            "value": "Results — Construction of crafted experiments"
          },
          "source_id": "crafted-experiments-to-evaluate-feature-selection-meth",
          "source_type": "work",
          "supports": [
            "/modalities"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "crafted-experiments-automated-metadata-8-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": "lqaf023",
            "source_fragment_sha256": "9c7d94ed56234955e3af590d0f71ddc95512d00d6172b590b4fef7441b3ed56d",
            "type": "section",
            "value": "Abstract"
          },
          "source_id": "crafted-experiments-to-evaluate-feature-selection-meth",
          "source_type": "work",
          "supports": [
            "/name"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "crafted-experiments-automated-metadata-9-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": "lqaf023",
            "source_fragment_sha256": "96a5e3c994e2c35a4d366970253271a7efa30b9765f7baf91b42ab2c85949723",
            "type": "section",
            "value": "Contributor Information"
          },
          "source_id": "crafted-experiments-to-evaluate-feature-selection-meth",
          "source_type": "work",
          "supports": [
            "/organizations"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "crafted-experiments-automated-metadata-10-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": "lqaf023",
            "source_fragment_sha256": "a700da6ffadac53812339f89289ff6a59f78616865a15f896489123e4caf11a6",
            "type": "section",
            "value": "Introduction"
          },
          "source_id": "crafted-experiments-to-evaluate-feature-selection-meth",
          "source_type": "work",
          "supports": [
            "/summary"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "crafted-experiments-automated-bibliographic-evidence",
          "locator": {
            "document_page": null,
            "note": "Resolved from the canonical paper identifier during intake.",
            "printed_page": null,
            "source_fragment_sha256": "937543b1533bf7b109e45d079b6772c1bc6b358b18d6eaabe89fd92ca1c28c77",
            "type": "other",
            "value": "Crossref bibliographic metadata"
          },
          "source_id": "crafted-experiments-to-evaluate-feature-selection-meth",
          "source_type": "work",
          "supports": [
            "/release_date"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "crafted-experiments-automated-count-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": "lqaf023",
            "source_fragment_sha256": "06f529272f7b8982461611c8ed44a3f2c517e8a83b90a853d44e8cb41445446f",
            "type": "section",
            "value": "Results — Construction of crafted experiments"
          },
          "source_id": "crafted-experiments-to-evaluate-feature-selection-meth",
          "source_type": "work",
          "supports": [
            "/task_counts/total",
            "/task_counts/basis",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "crafted-experiments-automated-subset-coverage-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": "lqaf023",
            "source_fragment_sha256": "06f529272f7b8982461611c8ed44a3f2c517e8a83b90a853d44e8cb41445446f",
            "type": "section",
            "value": "Results — Construction of crafted experiments"
          },
          "source_id": "crafted-experiments-to-evaluate-feature-selection-meth",
          "source_type": "work",
          "supports": [
            "/task_counts/subsets",
            "/versions/0/task_counts/subsets"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "crafted-experiments-automated-resource-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": "lqaf023",
            "source_fragment_sha256": "26c2b38fbd6d56cfb43907c46c68255b69de81a0bf5467213d28236efaad84ef",
            "type": "section",
            "value": "Data availability"
          },
          "source_id": "crafted-experiments-to-evaluate-feature-selection-meth",
          "source_type": "work",
          "supports": [
            "/resources",
            "/access/license"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "crafted-experiments-automated-creator-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": "lqaf023",
            "source_fragment_sha256": "eb2749dd0815c1edd91c79d8b811c1abfa8f4258a27f3cbfe973d38a35e76ba1",
            "type": "other",
            "value": "Article DOI"
          },
          "source_id": "crafted-experiments-to-evaluate-feature-selection-meth",
          "source_type": "work",
          "supports": [
            "/resources/0"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "crafted-experiments-automated-version-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": "lqaf023",
            "source_fragment_sha256": "eb2749dd0815c1edd91c79d8b811c1abfa8f4258a27f3cbfe973d38a35e76ba1",
            "type": "other",
            "value": "Article DOI"
          },
          "source_id": "crafted-experiments-to-evaluate-feature-selection-meth",
          "source_type": "work",
          "supports": [
            "/latest_version",
            "/versions/0"
          ]
        }
      ],
      "field_status": [
        {
          "confidence": "high",
          "evidence_ids": [
            "crafted-experiments-automated-subset-coverage-evidence"
          ],
          "path": "/task_counts/subsets",
          "reason": "The double-pass review established the root item total but did not establish an exhaustive formal-subset inventory; the empty list must not be interpreted as evidence that the benchmark has no subsets.",
          "status": "provisional"
        },
        {
          "confidence": "high",
          "evidence_ids": [
            "crafted-experiments-automated-resource-evidence"
          ],
          "path": "/access/license",
          "reason": "The double-pass review verified the official resource identity but did not establish a redistributable benchmark license; the value remains null pending source-level license verification.",
          "status": "provisional"
        }
      ],
      "id": "crafted-experiments",
      "implementations": [],
      "kind": "suite",
      "latest_version": "initial-release",
      "modalities": [
        "raw-omics"
      ],
      "name": "crafted experiments",
      "organizations": [
        "Lineberger Comprehensive Cancer Center, University of North Carolina",
        "University of North Carolina at Chapel Hill"
      ],
      "parent_id": null,
      "release_date": "2025-01-07",
      "resources": [
        {
          "access_notes": "Versioned creator source; full text is not mirrored.",
          "id": "crafted-experiments-creator-paper-resource",
          "last_checked": "2026-07-31",
          "license": null,
          "pin": null,
          "type": "paper",
          "url": "https://doi.org/10.1093/nargab/lqaf023"
        },
        {
          "access_notes": "Official resource pinned immutably during intake.",
          "id": "crafted-experiments-official-dataset-resource",
          "last_checked": "2026-07-31",
          "license": null,
          "pin": {
            "kind": "version",
            "url": "https://zenodo.org/records/13830885",
            "value": "v1.0.0"
          },
          "type": "dataset",
          "url": "https://zenodo.org/records/13830885"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "initial-release",
        "entries": [],
        "notes": "No Scientific Task claim passed independent high-confidence verification; task mapping remains pending a targeted official-source audit.",
        "status": "partial"
      },
      "summary": "Real single-cell RNA-seq data augmented with known gene perturbations for comparing feature-selection methods.",
      "task_counts": {
        "basis": "Explicitly reported generated inventory",
        "reporting_status": "reported",
        "subsets": [],
        "total": 24
      },
      "task_formats": [
        "unclassified"
      ],
      "verification": {
        "last_verified": "2026-07-31",
        "notes": "New family admitted after creator source and official resource verification; the unresolved benchmark license is visibly flagged.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": null,
          "evidence_ids": [
            "crafted-experiments-automated-count-evidence",
            "crafted-experiments-automated-version-evidence"
          ],
          "formal_tracks": [],
          "id": "crafted-experiments-initial-release-version",
          "label": "initial-release",
          "notes": "Atlas snapshot label for the creator-paper release; the source does not report a formal benchmark version.",
          "release_date": "2025-01-07",
          "status": "current",
          "task_counts": {
            "basis": "Explicitly reported generated inventory",
            "reporting_status": "reported",
            "subsets": [],
            "total": 24
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official repository publishes processed split archives, preprocessing notebooks, baseline code, and dataset-specific documentation.",
        "biosafety_notes": "AAV concerns capsid viability for gene-therapy research. This registry mirrors no sequences, measurements, or model outputs.",
        "grader": "Deterministic Spearman correlation and mean-squared-error calculations are implemented in the public baseline code.",
        "level": "fully-open",
        "license": "AFL-3.0 for FLIP code and derived task files; original GB1 data are CC BY 4.0; AAV and Meltome were obtained with creator permission.",
        "tasks": "All original AAV, GB1, and Meltome split CSVs and split definitions are public."
      },
      "aliases": [
        "Fitness Landscape Inference for Proteins"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Original FLIP, later repository additions, and FLIP2 are explicitly separated; landscape and active/discourse split counts were checked against the paper and pinned repository.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction",
        "regression"
      ],
      "coverage_notes": [
        {
          "count": 5,
          "coverage": "explicitly-in-scope",
          "notes": "All five GB1 dataset-split tasks predict the fitness of an immunoglobulin-binding protein domain; this is task count, not a distinct set of five assays.",
          "reporting_status": "reported",
          "tag": "protein-protein-binding"
        },
        {
          "count": null,
          "coverage": "explicitly-in-scope",
          "notes": "All tasks are motivated by protein engineering, but FLIP evaluates prediction/regression rather than sequence generation or optimization and does not publish a separate design-task count.",
          "reporting_status": "not_reported",
          "tag": "protein-design"
        }
      ],
      "domains": [
        "protein-sequence",
        "protein-design",
        "protein-protein-binding"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "flip-evidence-paper-definition",
          "locator": {
            "note": "Defines the three landscapes, 15 dataset splits, task inputs/outputs, baseline classes, sample counts, licenses, and protein-engineering motivation.",
            "type": "table",
            "value": "pp. 3–7, Figure 1 and Tables 2–3"
          },
          "source_id": "flip-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/organizations",
            "/release_date",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/versions/0/release_date",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0",
            "/scientific_task_classification/entries/1"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "flip-evidence-repository-splits",
          "locator": {
            "note": "Separates active, discourse-only, and obsolete splits; documents dataset-specific licensing and pins the runnable implementation.",
            "type": "repository-path",
            "value": "README.md; splits/README.md; splits/{aav,gb1,meltome}/README.md; LICENSE at commit 62cace8735f5610e2743cf06ce0f944b37fffaa6"
          },
          "source_id": "flip-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/subsets",
            "/resources",
            "/implementations",
            "/versions/0/as_of",
            "/versions/0/task_counts/subsets",
            "/versions/0/formal_tracks"
          ]
        }
      ],
      "field_status": [],
      "id": "flip",
      "implementations": [
        {
          "commit": "62cace8735f5610e2743cf06ce0f944b37fffaa6",
          "framework": "FLIP reference baselines",
          "notes": "Pinned official baseline training and scoring code; no original-2021 release tag exists in the repository.",
          "status": "official",
          "url": "https://github.com/J-SNACKKB/FLIP/tree/62cace8735f5610e2743cf06ce0f944b37fffaa6/baselines"
        }
      ],
      "kind": "suite",
      "latest_version": "original-2021",
      "modalities": [
        "protein-sequence",
        "table"
      ],
      "name": "FLIP",
      "organizations": [
        "Technical University of Munich",
        "Microsoft Research New England",
        "California Institute of Technology",
        "University of California Berkeley",
        "Massachusetts Institute of Technology",
        "Salesforce Research"
      ],
      "parent_id": null,
      "release_date": "2021-10-11",
      "resources": [
        {
          "access_notes": "Final NeurIPS 2021 Datasets and Benchmarks creator paper and supplementary material.",
          "id": "flip-paper-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY 4.0",
          "pin": {
            "kind": "snapshot",
            "url": "https://flip.protein.properties/assets/FLIP_2021_manuscript.pdf",
            "value": "sha256:afcf360c88a7a4ae153b3c2d8d4fa6d4ac0abe84f2131b94409abff6447ce363"
          },
          "type": "paper",
          "url": "https://flip.protein.properties/assets/FLIP_2021_manuscript.pdf"
        },
        {
          "access_notes": "Official living project site; it now distinguishes original FLIP legacy datasets from FLIP2.",
          "id": "flip-project-site-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "snapshot",
            "url": "https://flip.protein.properties/",
            "value": "2026-07-21"
          },
          "type": "website",
          "url": "https://flip.protein.properties/"
        },
        {
          "access_notes": "Official creator repository. The pinned current commit contains original FLIP paths and later additions; only aav, gb1, and meltome are counted in original-2021.",
          "id": "flip-repository-resource",
          "last_checked": "2026-07-21",
          "license": "AFL-3.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/J-SNACKKB/FLIP/tree/62cace8735f5610e2743cf06ce0f944b37fffaa6",
            "value": "62cace8735f5610e2743cf06ce0f944b37fffaa6"
          },
          "type": "repository",
          "url": "https://github.com/J-SNACKKB/FLIP"
        },
        {
          "access_notes": "Creator-hosted original benchmark data mirror linked by the paper and repository.",
          "id": "flip-dataset-resource",
          "last_checked": "2026-07-21",
          "license": "mixed",
          "pin": {
            "kind": "snapshot",
            "url": "http://data.bioembeddings.com/public/FLIP/",
            "value": "2026-07-21"
          },
          "type": "dataset",
          "url": "http://data.bioembeddings.com/public/FLIP/"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "original-2021",
        "entries": [
          {
            "confidence": "high",
            "count": 15,
            "count_basis": "Dataset-and-split benchmark tasks.",
            "count_ref": "/task_counts/total",
            "count_unit": "tasks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "flip-evidence-paper-definition"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "All formal tasks are supervised fitness or phenotype prediction.",
            "reporting_status": "reported",
            "task_type_id": "protein-fitness-prediction"
          },
          {
            "confidence": "high",
            "count": 15,
            "count_basis": "Dataset-and-split benchmark tasks.",
            "count_ref": "/task_counts/total",
            "count_unit": "tasks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "flip-evidence-paper-definition"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "The same tasks evaluate generalization across mutated sequence landscapes; claims overlap and are never summed.",
            "reporting_status": "reported",
            "task_type_id": "protein-mutation-effect-prediction"
          }
        ],
        "notes": "FLIP evaluates sequence-to-fitness prediction; it is not a sequence-generation benchmark.",
        "status": "complete"
      },
      "summary": "A supervised protein sequence-to-fitness benchmark that turns three experimental landscapes into 15 biologically motivated dataset splits for testing generalization in protein engineering.",
      "task_counts": {
        "basis": "dataset-and-split benchmark tasks",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "dataset-and-split tasks",
            "count": 7,
            "exclusive": true,
            "exhaustive": true,
            "id": "flip-aav-tasks",
            "label": "AAV landscape splits",
            "notes": "Six active comparison splits plus one sampled split used mainly for discourse.",
            "reporting_status": "reported"
          },
          {
            "basis": "dataset-and-split tasks",
            "count": 5,
            "exclusive": true,
            "exhaustive": true,
            "id": "flip-gb1-tasks",
            "label": "GB1 landscape splits",
            "notes": "Four active comparison splits plus one sampled split used mainly for discourse.",
            "reporting_status": "reported"
          },
          {
            "basis": "dataset-and-split tasks",
            "count": 3,
            "exclusive": true,
            "exhaustive": true,
            "id": "flip-meltome-tasks",
            "label": "Meltome thermostability splits",
            "notes": "Mixed, Human, and Human-cell are all active.",
            "reporting_status": "reported"
          },
          {
            "basis": "dataset-and-split tasks",
            "count": 13,
            "exclusive": false,
            "exhaustive": false,
            "id": "flip-active-comparison-tasks",
            "label": "Active performance-comparison splits",
            "notes": "Current official repository semaphore marks all original splits active except the two sampled splits.",
            "reporting_status": "reported"
          },
          {
            "basis": "dataset-and-split tasks",
            "count": 2,
            "exclusive": false,
            "exhaustive": false,
            "id": "flip-discourse-sampled-tasks",
            "label": "Sampled discourse splits",
            "notes": "AAV Sampled and GB1 Sampled are orange: the repository warns against performance comparisons because random sampling can overestimate performance.",
            "reporting_status": "reported"
          }
        ],
        "total": 15
      },
      "task_formats": [
        "supervised sequence-to-fitness regression"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Field-level family audit complete.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "flip-evidence-paper-definition",
            "flip-evidence-repository-splits"
          ],
          "formal_tracks": [
            "flip-aav",
            "flip-gb1",
            "flip-meltome"
          ],
          "id": "flip-original-2021",
          "label": "original-2021",
          "notes": "This record is intentionally limited to the three landscapes and 15 splits in the 2021 creator paper. Later repository datasets and FLIP2 are not folded into this version.",
          "release_date": "2021-10-11",
          "status": "current",
          "task_counts": {
            "basis": "dataset-and-split benchmark tasks",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "dataset-and-split tasks",
                "count": 7,
                "exclusive": true,
                "exhaustive": true,
                "id": "flip-aav-tasks",
                "label": "AAV landscape splits",
                "notes": "Six active comparison splits plus one sampled split used mainly for discourse.",
                "reporting_status": "reported"
              },
              {
                "basis": "dataset-and-split tasks",
                "count": 5,
                "exclusive": true,
                "exhaustive": true,
                "id": "flip-gb1-tasks",
                "label": "GB1 landscape splits",
                "notes": "Four active comparison splits plus one sampled split used mainly for discourse.",
                "reporting_status": "reported"
              },
              {
                "basis": "dataset-and-split tasks",
                "count": 3,
                "exclusive": true,
                "exhaustive": true,
                "id": "flip-meltome-tasks",
                "label": "Meltome thermostability splits",
                "notes": "Mixed, Human, and Human-cell are all active.",
                "reporting_status": "reported"
              },
              {
                "basis": "dataset-and-split tasks",
                "count": 13,
                "exclusive": false,
                "exhaustive": false,
                "id": "flip-active-comparison-tasks",
                "label": "Active performance-comparison splits",
                "notes": "Current official repository semaphore marks all original splits active except the two sampled splits.",
                "reporting_status": "reported"
              },
              {
                "basis": "dataset-and-split tasks",
                "count": 2,
                "exclusive": false,
                "exhaustive": false,
                "id": "flip-discourse-sampled-tasks",
                "label": "Sampled discourse splits",
                "notes": "AAV Sampled and GB1 Sampled are orange: the repository warns against performance comparisons because random sampling can overestimate performance.",
                "reporting_status": "reported"
              }
            ],
            "total": 15
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "Processed splits, raw-data preparation notebook, reference sequence, and baselines are public.",
        "biosafety_notes": "AAV2 capsid viability is relevant to gene-therapy vector engineering; this registry stores metadata and aggregate results only.",
        "grader": "Deterministic Spearman correlation and MSE.",
        "level": "fully-open",
        "license": "AFL-3.0 for modified split data; raw AAV source was obtained with written permission and its upstream repository has no explicit license.",
        "tasks": "All seven regression split CSVs are public."
      },
      "aliases": [
        "FLIP AAV capsid viability landscape"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "All train/test counts and AAV-specific protocol details were checked against the creator paper and pinned official split documentation.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction",
        "regression"
      ],
      "coverage_notes": [
        {
          "count": null,
          "coverage": "explicitly-in-scope",
          "notes": "The dataset includes model-designed variants and is motivated by capsid engineering, but the evaluated task is fitness prediction rather than sequence generation.",
          "reporting_status": "not_reported",
          "tag": "protein-design"
        }
      ],
      "domains": [
        "protein-sequence",
        "protein-design"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "flip-paper"
      ],
      "evaluation_run_ids": [
        "flip-aav-des-mut",
        "flip-aav-low-vs-high",
        "flip-aav-mut-des",
        "flip-aav-one-vs-rest",
        "flip-aav-sampled",
        "flip-aav-seven-vs-rest",
        "flip-aav-two-vs-rest"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "flip-aav-evidence-paper",
          "locator": {
            "note": "AAV definition, seven splits, exact train/test counts, regression setup, and baseline results.",
            "type": "table",
            "value": "pp. 4, 6, 9; Table 2, Section 3.2, Tables 5 and 7"
          },
          "source_id": "flip-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/summary",
            "/kind",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/versions/0/release_date",
            "/versions/0/as_of",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "flip-aav-evidence-repository",
          "locator": {
            "note": "Confirms regression/binary files, active versus sampled status, access, and AFL-3.0 derivative license.",
            "type": "repository-path",
            "value": "splits/aav/README.md and splits/aav/splits.zip at commit 62cace8735f5610e2743cf06ce0f944b37fffaa6"
          },
          "source_id": "flip-aav-repository-resource",
          "source_type": "resource",
          "supports": [
            "/access/tasks",
            "/access/artifacts",
            "/access/license",
            "/resources",
            "/implementations"
          ]
        }
      ],
      "field_status": [],
      "id": "flip-aav",
      "implementations": [
        {
          "commit": "62cace8735f5610e2743cf06ce0f944b37fffaa6",
          "framework": "FLIP AAV baselines",
          "notes": "Use dataset aav and the paper-to-repository split-name mappings.",
          "status": "official",
          "url": "https://github.com/J-SNACKKB/FLIP/tree/62cace8735f5610e2743cf06ce0f944b37fffaa6/baselines"
        }
      ],
      "kind": "track",
      "latest_version": "original-2021",
      "modalities": [
        "protein-sequence",
        "table"
      ],
      "name": "FLIP AAV",
      "organizations": [
        "Technical University of Munich",
        "Microsoft Research New England",
        "California Institute of Technology",
        "University of California Berkeley",
        "Massachusetts Institute of Technology",
        "Salesforce Research"
      ],
      "parent_id": "flip",
      "release_date": "2021-10-11",
      "resources": [
        {
          "access_notes": "Creator-authorized manuscript, especially Section 3.2 and Tables 2, 5, and 7.",
          "id": "flip-aav-paper-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY 4.0",
          "pin": {
            "kind": "snapshot",
            "url": "https://flip.protein.properties/assets/FLIP_2021_manuscript.pdf",
            "value": "sha256:afcf360c88a7a4ae153b3c2d8d4fa6d4ac0abe84f2131b94409abff6447ce363"
          },
          "type": "paper",
          "url": "https://flip.protein.properties/assets/FLIP_2021_manuscript.pdf"
        },
        {
          "access_notes": "Official AAV split archive and documentation.",
          "id": "flip-aav-repository-resource",
          "last_checked": "2026-07-21",
          "license": "AFL-3.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/J-SNACKKB/FLIP/tree/62cace8735f5610e2743cf06ce0f944b37fffaa6/splits/aav",
            "value": "62cace8735f5610e2743cf06ce0f944b37fffaa6"
          },
          "type": "repository",
          "url": "https://github.com/J-SNACKKB/FLIP/tree/62cace8735f5610e2743cf06ce0f944b37fffaa6/splits/aav"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "original-2021",
        "entries": [
          {
            "confidence": "high",
            "count": 284009,
            "count_basis": "Distinct sequence-fitness examples across the sampled and designed AAV pools.",
            "count_ref": "/task_counts/total",
            "count_unit": "examples",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "flip-aav-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Example count, not a count of formal split tasks.",
            "reporting_status": "reported",
            "task_type_id": "protein-fitness-prediction"
          }
        ],
        "notes": "AAV sequence-to-fitness landscape.",
        "status": "complete"
      },
      "summary": "Seven supervised splits over sampled and machine-designed AAV2 VP-1 capsid variants, measuring generalization across mutation depth, fitness, and sampled-versus-designed pools.",
      "task_counts": {
        "basis": "distinct sequence-fitness examples across the sampled and designed AAV pools",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "test examples",
            "count": 201426,
            "exclusive": false,
            "exhaustive": false,
            "id": "aav-mut-des-test",
            "label": "Mut-Des test set",
            "notes": "Train on 82,583 sampled variants; test on 201,426 designed variants.",
            "reporting_status": "reported"
          },
          {
            "basis": "test examples",
            "count": 82583,
            "exclusive": false,
            "exhaustive": false,
            "id": "aav-des-mut-test",
            "label": "Des-Mut test set",
            "notes": "Train on 201,426 designed variants; test on 82,583 sampled variants.",
            "reporting_status": "reported"
          },
          {
            "basis": "test examples",
            "count": 81413,
            "exclusive": false,
            "exhaustive": false,
            "id": "aav-one-vs-rest-test",
            "label": "1-vs-rest test set",
            "notes": "1,170 training examples; sampled pool only.",
            "reporting_status": "reported"
          },
          {
            "basis": "test examples",
            "count": 50776,
            "exclusive": false,
            "exhaustive": false,
            "id": "aav-two-vs-rest-test",
            "label": "2-vs-rest test set",
            "notes": "31,807 training examples; sampled pool only.",
            "reporting_status": "reported"
          },
          {
            "basis": "test examples",
            "count": 12581,
            "exclusive": false,
            "exhaustive": false,
            "id": "aav-seven-vs-rest-test",
            "label": "7-vs-rest test set",
            "notes": "70,002 training examples; sampled pool only.",
            "reporting_status": "reported"
          },
          {
            "basis": "test examples",
            "count": 35037,
            "exclusive": false,
            "exhaustive": false,
            "id": "aav-low-vs-high-test",
            "label": "low-vs-high test set",
            "notes": "47,546 training examples at or below wild-type fitness; test examples are above wild-type fitness.",
            "reporting_status": "reported"
          },
          {
            "basis": "test examples",
            "count": 16517,
            "exclusive": false,
            "exhaustive": false,
            "id": "aav-sampled-test",
            "label": "Sampled test set",
            "notes": "66,066 random training examples; discourse-only split not recommended for performance comparison.",
            "reporting_status": "reported"
          }
        ],
        "total": 284009
      },
      "task_formats": [
        "sequence-to-capsid-viability regression"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Field-level track audit complete.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "flip-aav-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "flip-aav-original-2021",
          "label": "original-2021",
          "notes": "The seven task splits overlap and therefore are not summed as sample partitions.",
          "release_date": "2021-10-11",
          "status": "current",
          "task_counts": {
            "basis": "distinct sequence-fitness examples across the sampled and designed AAV pools",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "test examples",
                "count": 201426,
                "exclusive": false,
                "exhaustive": false,
                "id": "aav-mut-des-test",
                "label": "Mut-Des test set",
                "notes": "Train on 82,583 sampled variants; test on 201,426 designed variants.",
                "reporting_status": "reported"
              },
              {
                "basis": "test examples",
                "count": 82583,
                "exclusive": false,
                "exhaustive": false,
                "id": "aav-des-mut-test",
                "label": "Des-Mut test set",
                "notes": "Train on 201,426 designed variants; test on 82,583 sampled variants.",
                "reporting_status": "reported"
              },
              {
                "basis": "test examples",
                "count": 81413,
                "exclusive": false,
                "exhaustive": false,
                "id": "aav-one-vs-rest-test",
                "label": "1-vs-rest test set",
                "notes": "1,170 training examples; sampled pool only.",
                "reporting_status": "reported"
              },
              {
                "basis": "test examples",
                "count": 50776,
                "exclusive": false,
                "exhaustive": false,
                "id": "aav-two-vs-rest-test",
                "label": "2-vs-rest test set",
                "notes": "31,807 training examples; sampled pool only.",
                "reporting_status": "reported"
              },
              {
                "basis": "test examples",
                "count": 12581,
                "exclusive": false,
                "exhaustive": false,
                "id": "aav-seven-vs-rest-test",
                "label": "7-vs-rest test set",
                "notes": "70,002 training examples; sampled pool only.",
                "reporting_status": "reported"
              },
              {
                "basis": "test examples",
                "count": 35037,
                "exclusive": false,
                "exhaustive": false,
                "id": "aav-low-vs-high-test",
                "label": "low-vs-high test set",
                "notes": "47,546 training examples at or below wild-type fitness; test examples are above wild-type fitness.",
                "reporting_status": "reported"
              },
              {
                "basis": "test examples",
                "count": 16517,
                "exclusive": false,
                "exhaustive": false,
                "id": "aav-sampled-test",
                "label": "Sampled test set",
                "notes": "66,066 random training examples; discourse-only split not recommended for performance comparison.",
                "reporting_status": "reported"
              }
            ],
            "total": 284009
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "Processed splits, original-data archive, reference sequence, and baselines are public.",
        "biosafety_notes": "No additional restriction is asserted; this registry stores metadata and aggregate results only.",
        "grader": "Deterministic Spearman correlation and MSE.",
        "level": "fully-open",
        "license": "Original Wu et al. GB1 data are CC BY 4.0; FLIP-modified task files are AFL-3.0.",
        "tasks": "All five regression split CSVs are public."
      },
      "aliases": [
        "FLIP protein G domain B1 landscape"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Downsampling, binding scope, train/test counts, and the sampled-split warning were verified.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction",
        "regression"
      ],
      "coverage_notes": [
        {
          "count": 8733,
          "coverage": "explicitly-in-scope",
          "notes": "Every retained GB1 example has an immunoglobulin-binding fitness value; this is an example count, while the parent suite reports five binding tasks.",
          "reporting_status": "reported",
          "tag": "protein-protein-binding"
        }
      ],
      "domains": [
        "protein-sequence",
        "protein-protein-binding",
        "protein-design"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "flip-paper"
      ],
      "evaluation_run_ids": [
        "flip-gb1-low-vs-high",
        "flip-gb1-one-vs-rest",
        "flip-gb1-sampled",
        "flip-gb1-three-vs-rest",
        "flip-gb1-two-vs-rest"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "flip-gb1-evidence-paper",
          "locator": {
            "note": "GB1 binding definition, downsampling to 8,733 examples, five splits, exact train/test counts, and results.",
            "type": "table",
            "value": "pp. 4–5, 8–9; Table 2, Section 3.1, Tables 4 and 7"
          },
          "source_id": "flip-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/summary",
            "/kind",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/versions/0/release_date",
            "/versions/0/as_of",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "flip-gb1-evidence-repository",
          "locator": {
            "note": "Confirms assay semantics, active versus sampled status, public data, and licenses.",
            "type": "repository-path",
            "value": "splits/gb1/README.md and splits/gb1/splits.zip at commit 62cace8735f5610e2743cf06ce0f944b37fffaa6"
          },
          "source_id": "flip-gb1-repository-resource",
          "source_type": "resource",
          "supports": [
            "/access/tasks",
            "/access/artifacts",
            "/access/license",
            "/resources",
            "/implementations"
          ]
        }
      ],
      "field_status": [],
      "id": "flip-gb1",
      "implementations": [
        {
          "commit": "62cace8735f5610e2743cf06ce0f944b37fffaa6",
          "framework": "FLIP GB1 baselines",
          "notes": "Use dataset gb1 and the documented split-name mappings.",
          "status": "official",
          "url": "https://github.com/J-SNACKKB/FLIP/tree/62cace8735f5610e2743cf06ce0f944b37fffaa6/baselines"
        }
      ],
      "kind": "track",
      "latest_version": "original-2021",
      "modalities": [
        "protein-sequence",
        "table"
      ],
      "name": "FLIP GB1",
      "organizations": [
        "Technical University of Munich",
        "Microsoft Research New England",
        "California Institute of Technology",
        "University of California Berkeley",
        "Massachusetts Institute of Technology",
        "Salesforce Research"
      ],
      "parent_id": "flip",
      "release_date": "2021-10-11",
      "resources": [
        {
          "access_notes": "Creator-authorized manuscript, especially Section 3.1 and Tables 2, 4, and 7.",
          "id": "flip-gb1-paper-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY 4.0",
          "pin": {
            "kind": "snapshot",
            "url": "https://flip.protein.properties/assets/FLIP_2021_manuscript.pdf",
            "value": "sha256:afcf360c88a7a4ae153b3c2d8d4fa6d4ac0abe84f2131b94409abff6447ce363"
          },
          "type": "paper",
          "url": "https://flip.protein.properties/assets/FLIP_2021_manuscript.pdf"
        },
        {
          "access_notes": "Official GB1 split archive and documentation.",
          "id": "flip-gb1-repository-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY 4.0 raw data; AFL-3.0 derivatives",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/J-SNACKKB/FLIP/tree/62cace8735f5610e2743cf06ce0f944b37fffaa6/splits/gb1",
            "value": "62cace8735f5610e2743cf06ce0f944b37fffaa6"
          },
          "type": "repository",
          "url": "https://github.com/J-SNACKKB/FLIP/tree/62cace8735f5610e2743cf06ce0f944b37fffaa6/splits/gb1"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "original-2021",
        "entries": [
          {
            "confidence": "high",
            "count": 8733,
            "count_basis": "Downsampled sequence-fitness examples retained for FLIP.",
            "count_ref": "/task_counts/total",
            "count_unit": "examples",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "flip-gb1-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Binding context does not turn the task into direct affinity prediction.",
            "reporting_status": "reported",
            "task_type_id": "protein-fitness-prediction"
          }
        ],
        "notes": "GB1 binding-fitness landscape; the target is fitness rather than direct affinity estimation.",
        "status": "complete"
      },
      "summary": "Five supervised splits over a downsampled, highly epistatic four-site GB1 immunoglobulin-binding landscape, designed to test mutation-depth and low-to-high-fitness generalization.",
      "task_counts": {
        "basis": "downsampled sequence-fitness examples retained for FLIP",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "test examples",
            "count": 8704,
            "exclusive": false,
            "exhaustive": false,
            "id": "gb1-one-vs-rest-test",
            "label": "1-vs-rest test set",
            "notes": "29 training examples: wild type and single mutants.",
            "reporting_status": "reported"
          },
          {
            "basis": "test examples",
            "count": 8306,
            "exclusive": false,
            "exhaustive": false,
            "id": "gb1-two-vs-rest-test",
            "label": "2-vs-rest test set",
            "notes": "427 training examples: wild type, single, and double mutants.",
            "reporting_status": "reported"
          },
          {
            "basis": "test examples",
            "count": 5765,
            "exclusive": false,
            "exhaustive": false,
            "id": "gb1-three-vs-rest-test",
            "label": "3-vs-rest test set",
            "notes": "2,968 training examples through triple mutants.",
            "reporting_status": "reported"
          },
          {
            "basis": "test examples",
            "count": 3644,
            "exclusive": false,
            "exhaustive": false,
            "id": "gb1-low-vs-high-test",
            "label": "low-vs-high test set",
            "notes": "5,089 training examples at or below wild-type fitness.",
            "reporting_status": "reported"
          },
          {
            "basis": "test examples",
            "count": 1772,
            "exclusive": false,
            "exhaustive": false,
            "id": "gb1-sampled-test",
            "label": "Sampled test set",
            "notes": "6,961 random training examples; discourse-only split not recommended for performance comparison.",
            "reporting_status": "reported"
          }
        ],
        "total": 8733
      },
      "task_formats": [
        "sequence-to-immunoglobulin-binding-fitness regression"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Field-level track audit complete.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "flip-gb1-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "flip-gb1-original-2021",
          "label": "original-2021",
          "notes": "The five task splits overlap and therefore are not summed as sample partitions.",
          "release_date": "2021-10-11",
          "status": "current",
          "task_counts": {
            "basis": "downsampled sequence-fitness examples retained for FLIP",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "test examples",
                "count": 8704,
                "exclusive": false,
                "exhaustive": false,
                "id": "gb1-one-vs-rest-test",
                "label": "1-vs-rest test set",
                "notes": "29 training examples: wild type and single mutants.",
                "reporting_status": "reported"
              },
              {
                "basis": "test examples",
                "count": 8306,
                "exclusive": false,
                "exhaustive": false,
                "id": "gb1-two-vs-rest-test",
                "label": "2-vs-rest test set",
                "notes": "427 training examples: wild type, single, and double mutants.",
                "reporting_status": "reported"
              },
              {
                "basis": "test examples",
                "count": 5765,
                "exclusive": false,
                "exhaustive": false,
                "id": "gb1-three-vs-rest-test",
                "label": "3-vs-rest test set",
                "notes": "2,968 training examples through triple mutants.",
                "reporting_status": "reported"
              },
              {
                "basis": "test examples",
                "count": 3644,
                "exclusive": false,
                "exhaustive": false,
                "id": "gb1-low-vs-high-test",
                "label": "low-vs-high test set",
                "notes": "5,089 training examples at or below wild-type fitness.",
                "reporting_status": "reported"
              },
              {
                "basis": "test examples",
                "count": 1772,
                "exclusive": false,
                "exhaustive": false,
                "id": "gb1-sampled-test",
                "label": "Sampled test set",
                "notes": "6,961 random training examples; discourse-only split not recommended for performance comparison.",
                "reporting_status": "reported"
              }
            ],
            "total": 8733
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "Processed splits, Meltome source archive, FASTA, clustering file, and baselines are public.",
        "biosafety_notes": "No additional restriction is asserted; this registry stores metadata and aggregate results only.",
        "grader": "Deterministic Spearman correlation and MSE.",
        "level": "fully-open",
        "license": "Meltome creators state the original data are free to use with acknowledgement; FLIP-modified task files are AFL-3.0.",
        "tasks": "All three active regression split CSVs are public."
      },
      "aliases": [
        "FLIP Thermostability"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "The paper-versus-CSV Human-cell total conflict is resolved by the higher-priority commit-pinned official dataset: 7,158.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction",
        "regression"
      ],
      "coverage_notes": [
        {
          "count": 27951,
          "coverage": "explicitly-in-scope",
          "notes": "The benchmark uses Meltome Atlas thermal-proteome measurements; count is the Mixed split universe, not independent assays.",
          "reporting_status": "reported",
          "tag": "proteomics"
        }
      ],
      "domains": [
        "protein-sequence",
        "protein-design",
        "proteomics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "flip-paper"
      ],
      "evaluation_run_ids": [
        "flip-meltome-human",
        "flip-meltome-human-cell",
        "flip-meltome-mixed"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "flip-meltome-evidence-paper",
          "locator": {
            "note": "Defines three splits, train/test counts, 20%-identity clustering, metric, and baseline results; Table 2 prints Human-cell total 7,156.",
            "type": "table",
            "value": "pp. 4, 6, 9; Table 2, Section 3.3, Table 6"
          },
          "source_id": "flip-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/summary",
            "/kind",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/versions/0/release_date",
            "/versions/0/as_of",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "flip-meltome-evidence-repository",
          "locator": {
            "note": "CSV counts: Mixed 27,951=24,817+3,134; Human 10,093=8,148+1,945; Human-cell 7,158=5,792+1,366.",
            "type": "repository-path",
            "value": "splits/meltome/README.md and splits/meltome/splits.zip at commit 62cace8735f5610e2743cf06ce0f944b37fffaa6"
          },
          "source_id": "flip-meltome-repository-resource",
          "source_type": "resource",
          "supports": [
            "/task_counts/subsets",
            "/access/tasks",
            "/access/artifacts",
            "/access/license",
            "/resources",
            "/implementations",
            "/versions/0/task_counts/subsets"
          ]
        }
      ],
      "field_status": [],
      "id": "flip-meltome",
      "implementations": [
        {
          "commit": "62cace8735f5610e2743cf06ce0f944b37fffaa6",
          "framework": "FLIP Meltome baselines",
          "notes": "Use dataset meltome with mixed, human, or human_cell.",
          "status": "official",
          "url": "https://github.com/J-SNACKKB/FLIP/tree/62cace8735f5610e2743cf06ce0f944b37fffaa6/baselines"
        }
      ],
      "kind": "track",
      "latest_version": "original-2021",
      "modalities": [
        "protein-sequence",
        "table"
      ],
      "name": "FLIP Meltome Thermostability",
      "organizations": [
        "Technical University of Munich",
        "Microsoft Research New England",
        "California Institute of Technology",
        "University of California Berkeley",
        "Massachusetts Institute of Technology",
        "Salesforce Research"
      ],
      "parent_id": "flip",
      "release_date": "2021-10-11",
      "resources": [
        {
          "access_notes": "Creator-authorized manuscript, especially Section 3.3 and Tables 2 and 6.",
          "id": "flip-meltome-paper-resource",
          "last_checked": "2026-07-21",
          "license": "CC BY 4.0",
          "pin": {
            "kind": "snapshot",
            "url": "https://flip.protein.properties/assets/FLIP_2021_manuscript.pdf",
            "value": "sha256:afcf360c88a7a4ae153b3c2d8d4fa6d4ac0abe84f2131b94409abff6447ce363"
          },
          "type": "paper",
          "url": "https://flip.protein.properties/assets/FLIP_2021_manuscript.pdf"
        },
        {
          "access_notes": "Official Meltome split archive and documentation; the pinned CSV row counts resolve the Table 2 total mismatch.",
          "id": "flip-meltome-repository-resource",
          "last_checked": "2026-07-21",
          "license": "AFL-3.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/J-SNACKKB/FLIP/tree/62cace8735f5610e2743cf06ce0f944b37fffaa6/splits/meltome",
            "value": "62cace8735f5610e2743cf06ce0f944b37fffaa6"
          },
          "type": "repository",
          "url": "https://github.com/J-SNACKKB/FLIP/tree/62cace8735f5610e2743cf06ce0f944b37fffaa6/splits/meltome"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "original-2021",
        "entries": [
          {
            "confidence": "high",
            "count": 27951,
            "count_basis": "Sequence-temperature examples in the Mixed split universe.",
            "count_ref": "/task_counts/total",
            "count_unit": "examples",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "flip-meltome-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Temperature examples are not combined with AAV or GB1 counts.",
            "reporting_status": "reported",
            "task_type_id": "protein-stability-prediction"
          }
        ],
        "notes": "Protein melting-temperature landscape.",
        "status": "complete"
      },
      "summary": "Three supervised sequence-to-melting-temperature splits spanning all species, human proteins, and a single human cell line, with sequence-cluster-aware train/test separation.",
      "task_counts": {
        "basis": "sequence-temperature examples in the Mixed split universe",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "test examples",
            "count": 3134,
            "exclusive": false,
            "exhaustive": false,
            "id": "meltome-mixed-test",
            "label": "Mixed test set",
            "notes": "24,817 training examples; test contains representatives from held-out 20%-identity clusters.",
            "reporting_status": "reported"
          },
          {
            "basis": "sequence-temperature examples",
            "count": 10093,
            "exclusive": false,
            "exhaustive": false,
            "id": "meltome-human-all",
            "label": "Human split total",
            "notes": "Human-only subset of the landscape.",
            "reporting_status": "reported"
          },
          {
            "basis": "test examples",
            "count": 1945,
            "exclusive": false,
            "exhaustive": false,
            "id": "meltome-human-test",
            "label": "Human test set",
            "notes": "8,148 training examples.",
            "reporting_status": "reported"
          },
          {
            "basis": "sequence-temperature examples",
            "count": 7158,
            "exclusive": false,
            "exhaustive": false,
            "id": "meltome-human-cell-all",
            "label": "Human-cell split total",
            "notes": "Pinned official CSV has 7,158 rows (5,792 train + 1,366 test); paper Table 2 prints 7,156.",
            "reporting_status": "reported"
          },
          {
            "basis": "test examples",
            "count": 1366,
            "exclusive": false,
            "exhaustive": false,
            "id": "meltome-human-cell-test",
            "label": "Human-cell test set",
            "notes": "5,792 training examples in the pinned CSV and paper.",
            "reporting_status": "reported"
          }
        ],
        "total": 27951
      },
      "task_formats": [
        "sequence-to-melting-temperature regression"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Field-level track audit complete.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "flip-meltome-evidence-paper",
            "flip-meltome-evidence-repository"
          ],
          "formal_tracks": [],
          "id": "flip-meltome-original-2021",
          "label": "original-2021",
          "notes": "The versioned repository CSV is authoritative for the Human-cell total; the creator-paper discrepancy remains explicit in notes and evidence.",
          "release_date": "2021-10-11",
          "status": "current",
          "task_counts": {
            "basis": "sequence-temperature examples in the Mixed split universe",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "test examples",
                "count": 3134,
                "exclusive": false,
                "exhaustive": false,
                "id": "meltome-mixed-test",
                "label": "Mixed test set",
                "notes": "24,817 training examples; test contains representatives from held-out 20%-identity clusters.",
                "reporting_status": "reported"
              },
              {
                "basis": "sequence-temperature examples",
                "count": 10093,
                "exclusive": false,
                "exhaustive": false,
                "id": "meltome-human-all",
                "label": "Human split total",
                "notes": "Human-only subset of the landscape.",
                "reporting_status": "reported"
              },
              {
                "basis": "test examples",
                "count": 1945,
                "exclusive": false,
                "exhaustive": false,
                "id": "meltome-human-test",
                "label": "Human test set",
                "notes": "8,148 training examples.",
                "reporting_status": "reported"
              },
              {
                "basis": "sequence-temperature examples",
                "count": 7158,
                "exclusive": false,
                "exhaustive": false,
                "id": "meltome-human-cell-all",
                "label": "Human-cell split total",
                "notes": "Pinned official CSV has 7,158 rows (5,792 train + 1,366 test); paper Table 2 prints 7,156.",
                "reporting_status": "reported"
              },
              {
                "basis": "test examples",
                "count": 1366,
                "exclusive": false,
                "exhaustive": false,
                "id": "meltome-human-cell-test",
                "label": "Human-cell test set",
                "notes": "5,792 training examples in the pinned CSV and paper.",
                "reporting_status": "reported"
              }
            ],
            "total": 27951
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The ten public packages include solver prompts, staged data, ground truth, problem-specific grader configuration, checksums, and detailed case-study reports.",
        "biosafety_notes": "Problems use constructively simulated data and synthetic labels. BioBench Atlas stores metadata and aggregate results only and does not mirror task data.",
        "grader": "A Python reference grader is public for the ten released case studies; the full 129-problem evaluation packages and grader targets remain held out.",
        "level": "partially-open",
        "license": "MIT (root LICENSE); CC-BY-4.0 (Hugging Face dataset-card metadata)",
        "tasks": "Ten externally reviewed problems are fully public; a disjoint 50-problem Artificial Analysis subset and 69-problem internal holdout are not public."
      },
      "aliases": [
        "Gene Bench Pro"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Counts, release strata, domain atlas, environment, grading, repeats, and all 60 configuration results are audited. The public package exposes conflicting license declarations.",
        "status": "audited-with-caveats",
        "unresolved_fields": 1
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "data-analysis",
        "coding",
        "tool-use",
        "scientific-reasoning"
      ],
      "coverage_notes": [
        {
          "count": 7,
          "coverage": "explicitly-in-scope",
          "notes": "Seven primary-domain problems are assigned to Proteomics; proteomics also appears in cross-domain problems.",
          "reporting_status": "reported",
          "tag": "proteomics"
        },
        {
          "count": null,
          "coverage": "unknown",
          "notes": "The domain atlas covers proteomics and biomarkers but does not define or count protein-binding problems separately.",
          "reporting_status": "not_reported",
          "tag": "protein-protein-binding"
        }
      ],
      "domains": [
        "genomics",
        "transcriptomics",
        "epigenomics",
        "single-cell",
        "spatial-omics",
        "proteomics",
        "microbiome",
        "clinical-translational",
        "bioinformatics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "genebench-pro-report"
      ],
      "evaluation_run_ids": [
        "genebench-pro-claude-high",
        "genebench-pro-claude-low",
        "genebench-pro-claude-max",
        "genebench-pro-claude-medium",
        "genebench-pro-claude-xhigh",
        "genebench-pro-official",
        "genebench-pro-pro-mode",
        "genebench-pro-reasoning-enabled",
        "genebench-pro-standard-high",
        "genebench-pro-standard-low",
        "genebench-pro-standard-max",
        "genebench-pro-standard-medium",
        "genebench-pro-standard-none"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "genebench-pro-paper-evidence",
          "locator": {
            "note": "Definition, 129 total, 10/50/69 release partition, 10 domains and 21 subdomains, 82/47 review strata, environment, grading, attempts, metrics, and results.",
            "type": "section",
            "value": "Abstract; Benchmark Scope and Construction; Construction, Validation, and Grading; Methods; Supplementary Tables 1–2"
          },
          "source_id": "genebench-pro-report",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/biosafety_notes",
            "/resources",
            "/implementations",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0",
            "/scientific_task_classification/entries/1"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "genebench-pro-public-package-evidence",
          "locator": {
            "note": "Ten public packages, released artifacts, runner contract, commit pin, and the CC-BY-4.0 versus MIT license conflict.",
            "type": "repository-path",
            "value": "README.md; LICENSE; problems.csv; manifest.json; reference_grader.py at 9bd2c54a6c0beef041e3504aa7eb65fc77783e18"
          },
          "source_id": "genebench-pro-public-dataset-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/implementations",
            "/versions/0/task_counts/subsets"
          ]
        }
      ],
      "field_status": [
        {
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-public-package-evidence"
          ],
          "path": "/access/license",
          "reason": "The pinned Hugging Face README front matter declares CC-BY-4.0, while the same package root LICENSE is the MIT License with OpenAI copyright.",
          "status": "conflicted"
        }
      ],
      "id": "genebench-pro",
      "implementations": [
        {
          "commit": "9bd2c54a6c0beef041e3504aa7eb65fc77783e18",
          "framework": "GeneBench-Pro public reference grader",
          "notes": "Python 3.10+ grader for the ten public reproducibility packages; the boolean passed field is authoritative.",
          "status": "official",
          "url": "https://huggingface.co/datasets/ajh-oai/genebench-pro-public-package/blob/9bd2c54a6c0beef041e3504aa7eb65fc77783e18/reference_grader.py"
        },
        {
          "commit": null,
          "framework": "GeneBench-Pro internal Docker evaluation harness",
          "notes": "The paper specifies the environment and protocol, but the full 129-problem runner, prompts, targets, and grader packages are not public.",
          "status": "not-available",
          "url": null
        }
      ],
      "kind": "agentic-eval",
      "latest_version": "paper-v1",
      "modalities": [
        "text",
        "table",
        "raw-omics",
        "code"
      ],
      "name": "GeneBench-Pro",
      "organizations": [
        "OpenAI"
      ],
      "parent_id": null,
      "release_date": "2026-06-30",
      "resources": [
        {
          "access_notes": "Official OpenAI PDF, SHA256 b1131c0c9b43598400ed2f2de40666b6587a11de757857fd43a33b46da9f2ba9.",
          "id": "genebench-pro-paper-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-NC-4.0 (bioRxiv v1)",
          "pin": {
            "kind": "version",
            "url": "https://www.biorxiv.org/content/10.64898/2026.06.29.735386v1",
            "value": "bioRxiv v1"
          },
          "type": "paper",
          "url": "https://cdn.openai.com/pdf/21938268-21af-442f-af93-3b2249afb241/genebench-pro.pdf"
        },
        {
          "access_notes": "Official launch page and interactive domain atlas.",
          "id": "genebench-pro-announcement-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "snapshot",
            "url": "https://openai.com/index/introducing-genebench-pro/",
            "value": "2026-07-21"
          },
          "type": "website",
          "url": "https://openai.com/index/introducing-genebench-pro/"
        },
        {
          "access_notes": "Official browser for the ten public case studies.",
          "id": "genebench-pro-case-studies-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "snapshot",
            "url": "https://openai.com/index/genebench-pro/case-studies/",
            "value": "2026-07-21"
          },
          "type": "website",
          "url": "https://openai.com/index/genebench-pro/case-studies/"
        },
        {
          "access_notes": "Official ten-problem public package. Dataset metadata and root LICENSE disagree on the package license.",
          "id": "genebench-pro-public-dataset-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-4.0 metadata; MIT root LICENSE",
          "pin": {
            "kind": "commit",
            "url": "https://huggingface.co/datasets/ajh-oai/genebench-pro-public-package/tree/9bd2c54a6c0beef041e3504aa7eb65fc77783e18",
            "value": "9bd2c54a6c0beef041e3504aa7eb65fc77783e18"
          },
          "type": "dataset",
          "url": "https://huggingface.co/datasets/ajh-oai/genebench-pro-public-package"
        },
        {
          "access_notes": "Reference implementation for the ten public case-study grader contracts.",
          "id": "genebench-pro-reference-grader-resource",
          "last_checked": "2026-07-21",
          "license": "MIT root LICENSE; CC-BY-4.0 dataset metadata",
          "pin": {
            "kind": "commit",
            "url": "https://huggingface.co/datasets/ajh-oai/genebench-pro-public-package/blob/9bd2c54a6c0beef041e3504aa7eb65fc77783e18/reference_grader.py",
            "value": "9bd2c54a6c0beef041e3504aa7eb65fc77783e18"
          },
          "type": "grader",
          "url": "https://huggingface.co/datasets/ajh-oai/genebench-pro-public-package/blob/9bd2c54a6c0beef041e3504aa7eb65fc77783e18/reference_grader.py"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "paper-v1",
        "entries": [
          {
            "confidence": "high",
            "count": 129,
            "count_basis": "self-contained synthetic scientific-analysis problems (called evaluations in the paper abstract)",
            "count_ref": "/task_counts/total",
            "count_unit": "problems",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "genebench-pro-paper-evidence"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "Problems require multistage analysis, diagnostics, and judgment.",
            "reporting_status": "reported",
            "task_type_id": "end-to-end-computational-analysis"
          },
          {
            "confidence": "high",
            "count": null,
            "count_basis": "Problems assigned to transcriptomics, epigenomics, single-cell, spatial, proteomics, microbiome, and related official domains.",
            "count_ref": null,
            "count_unit": "problems",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "genebench-pro-paper-evidence"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "The official domain atlas supports broad omics coverage but not an exhaustive leaf-task subtotal.",
            "reporting_status": "not_reported",
            "task_type_id": "omics-cellular-analysis"
          }
        ],
        "notes": "Official primary domains and task styles are not an exhaustive scientific-task taxonomy.",
        "status": "partial"
      },
      "summary": "A research-level agent benchmark of 129 synthetic, multistage computational-biology analyses that require iterative QC, statistical modeling, diagnostics, and decision-relevant judgment.",
      "task_counts": {
        "basis": "self-contained synthetic scientific-analysis problems (called evaluations in the paper abstract)",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "problems",
            "count": 10,
            "exclusive": true,
            "exhaustive": true,
            "id": "genebench-pro-public-release",
            "label": "Public release subset",
            "notes": "Ten externally reviewed case-study packages are public with prompts, staged data, ground truth, grader configuration, and reports.",
            "reporting_status": "reported"
          },
          {
            "basis": "held-out problems",
            "count": 50,
            "exclusive": true,
            "exhaustive": true,
            "id": "genebench-pro-artificial-analysis",
            "label": "Artificial Analysis reporting subset",
            "notes": "Disjoint from the public release. The formal paper says this locked reporting subset was provided to Artificial Analysis; the launch page still describes delivery in future tense. The tasks are not public.",
            "reporting_status": "reported"
          },
          {
            "basis": "held-out problems",
            "count": 69,
            "exclusive": true,
            "exhaustive": true,
            "id": "genebench-pro-internal-holdout",
            "label": "Internal holdout",
            "notes": "Remainder after the disjoint 10-problem public and 50-problem Artificial Analysis subsets; not publicly released.",
            "reporting_status": "reported"
          },
          {
            "basis": "problems in the primary-domain atlas",
            "count": 17,
            "exclusive": false,
            "exhaustive": false,
            "id": "genebench-pro-primary-statistical-genetics",
            "label": "Primary domain — Statistical genetics",
            "notes": "One member of the 10-domain partition; marked non-exhaustive here because the registry also stores independent release and review partitions.",
            "reporting_status": "reported"
          },
          {
            "basis": "problems in the primary-domain atlas",
            "count": 21,
            "exclusive": false,
            "exhaustive": false,
            "id": "genebench-pro-primary-population-genetics",
            "label": "Primary domain — Population genetics",
            "notes": "One member of the 10-domain partition; marked non-exhaustive here because the registry also stores independent release and review partitions.",
            "reporting_status": "reported"
          },
          {
            "basis": "problems in the primary-domain atlas",
            "count": 17,
            "exclusive": false,
            "exhaustive": false,
            "id": "genebench-pro-primary-quantitative-genetics",
            "label": "Primary domain — Quantitative genetics",
            "notes": "One member of the 10-domain partition; marked non-exhaustive here because the registry also stores independent release and review partitions.",
            "reporting_status": "reported"
          },
          {
            "basis": "problems in the primary-domain atlas",
            "count": 17,
            "exclusive": false,
            "exhaustive": false,
            "id": "genebench-pro-primary-regulatory-omics",
            "label": "Primary domain — Regulatory omics",
            "notes": "One member of the 10-domain partition; marked non-exhaustive here because the registry also stores independent release and review partitions.",
            "reporting_status": "reported"
          },
          {
            "basis": "problems in the primary-domain atlas",
            "count": 9,
            "exclusive": false,
            "exhaustive": false,
            "id": "genebench-pro-primary-functional-genomics",
            "label": "Primary domain — Functional genomics",
            "notes": "One member of the 10-domain partition; marked non-exhaustive here because the registry also stores independent release and review partitions.",
            "reporting_status": "reported"
          },
          {
            "basis": "problems in the primary-domain atlas",
            "count": 7,
            "exclusive": false,
            "exhaustive": false,
            "id": "genebench-pro-primary-proteomics",
            "label": "Primary domain — Proteomics",
            "notes": "One member of the 10-domain partition; marked non-exhaustive here because the registry also stores independent release and review partitions.",
            "reporting_status": "reported"
          },
          {
            "basis": "problems in the primary-domain atlas",
            "count": 26,
            "exclusive": false,
            "exhaustive": false,
            "id": "genebench-pro-primary-clinical-pgx-diagnostics",
            "label": "Primary domain — Clinical, PGx & diagnostics",
            "notes": "One member of the 10-domain partition; marked non-exhaustive here because the registry also stores independent release and review partitions.",
            "reporting_status": "reported"
          },
          {
            "basis": "problems in the primary-domain atlas",
            "count": 10,
            "exclusive": false,
            "exhaustive": false,
            "id": "genebench-pro-primary-cancer-genomics",
            "label": "Primary domain — Cancer genomics",
            "notes": "One member of the 10-domain partition; marked non-exhaustive here because the registry also stores independent release and review partitions.",
            "reporting_status": "reported"
          },
          {
            "basis": "problems in the primary-domain atlas",
            "count": 3,
            "exclusive": false,
            "exhaustive": false,
            "id": "genebench-pro-primary-microbial-genomics",
            "label": "Primary domain — Microbial genomics",
            "notes": "One member of the 10-domain partition; marked non-exhaustive here because the registry also stores independent release and review partitions.",
            "reporting_status": "reported"
          },
          {
            "basis": "problems in the primary-domain atlas",
            "count": 2,
            "exclusive": false,
            "exhaustive": false,
            "id": "genebench-pro-primary-forensic-genetics",
            "label": "Primary domain — Forensic genetics",
            "notes": "One member of the 10-domain partition; marked non-exhaustive here because the registry also stores independent release and review partitions.",
            "reporting_status": "reported"
          },
          {
            "basis": "problems in the terminal-subdomain atlas",
            "count": 6,
            "exclusive": false,
            "exhaustive": false,
            "id": "genebench-pro-terminal-association-correction",
            "label": "Terminal subdomain — Association & correction",
            "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
            "reporting_status": "reported"
          },
          {
            "basis": "problems in the terminal-subdomain atlas",
            "count": 6,
            "exclusive": false,
            "exhaustive": false,
            "id": "genebench-pro-terminal-causal-mapping",
            "label": "Terminal subdomain — Causal mapping",
            "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
            "reporting_status": "reported"
          },
          {
            "basis": "problems in the terminal-subdomain atlas",
            "count": 2,
            "exclusive": false,
            "exhaustive": false,
            "id": "genebench-pro-terminal-heritability-architecture",
            "label": "Terminal subdomain — Heritability and architecture",
            "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
            "reporting_status": "reported"
          },
          {
            "basis": "problems in the terminal-subdomain atlas",
            "count": 3,
            "exclusive": false,
            "exhaustive": false,
            "id": "genebench-pro-terminal-pedigree-ibd-phasing",
            "label": "Terminal subdomain — Pedigree, IBD, and phasing",
            "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
            "reporting_status": "reported"
          },
          {
            "basis": "problems in the terminal-subdomain atlas",
            "count": 7,
            "exclusive": false,
            "exhaustive": false,
            "id": "genebench-pro-terminal-selection-mutation",
            "label": "Terminal subdomain — Selection & mutation",
            "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
            "reporting_status": "reported"
          },
          {
            "basis": "problems in the terminal-subdomain atlas",
            "count": 6,
            "exclusive": false,
            "exhaustive": false,
            "id": "genebench-pro-terminal-admixture-adna",
            "label": "Terminal subdomain — Admixture & ancient DNA",
            "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
            "reporting_status": "reported"
          },
          {
            "basis": "problems in the terminal-subdomain atlas",
            "count": 8,
            "exclusive": false,
            "exhaustive": false,
            "id": "genebench-pro-terminal-history-genealogies",
            "label": "Terminal subdomain — History & genealogies",
            "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
            "reporting_status": "reported"
          },
          {
            "basis": "problems in the terminal-subdomain atlas",
            "count": 6,
            "exclusive": false,
            "exhaustive": false,
            "id": "genebench-pro-terminal-trait-architecture-variance",
            "label": "Terminal subdomain — Trait architecture and variance",
            "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
            "reporting_status": "reported"
          },
          {
            "basis": "problems in the terminal-subdomain atlas",
            "count": 6,
            "exclusive": false,
            "exhaustive": false,
            "id": "genebench-pro-terminal-family-social-transmission",
            "label": "Terminal subdomain — Family, social, and transmission effects",
            "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
            "reporting_status": "reported"
          },
          {
            "basis": "problems in the terminal-subdomain atlas",
            "count": 5,
            "exclusive": false,
            "exhaustive": false,
            "id": "genebench-pro-terminal-polygenic-prediction-selection",
            "label": "Terminal subdomain — Polygenic prediction and genomic selection",
            "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
            "reporting_status": "reported"
          },
          {
            "basis": "problems in the terminal-subdomain atlas",
            "count": 8,
            "exclusive": false,
            "exhaustive": false,
            "id": "genebench-pro-terminal-regulatory-qtls-ase",
            "label": "Terminal subdomain — Regulatory QTLs & ASE",
            "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
            "reporting_status": "reported"
          },
          {
            "basis": "problems in the terminal-subdomain atlas",
            "count": 5,
            "exclusive": false,
            "exhaustive": false,
            "id": "genebench-pro-terminal-transcriptome-structure",
            "label": "Terminal subdomain — Transcriptome structure",
            "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
            "reporting_status": "reported"
          },
          {
            "basis": "problems in the terminal-subdomain atlas",
            "count": 4,
            "exclusive": false,
            "exhaustive": false,
            "id": "genebench-pro-terminal-spatial-chromatin-context",
            "label": "Terminal subdomain — Spatial and chromatin context",
            "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
            "reporting_status": "reported"
          },
          {
            "basis": "problems in the terminal-subdomain atlas",
            "count": 9,
            "exclusive": false,
            "exhaustive": false,
            "id": "genebench-pro-terminal-functional-genomics-terminal",
            "label": "Terminal subdomain — Functional genomics",
            "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
            "reporting_status": "reported"
          },
          {
            "basis": "problems in the terminal-subdomain atlas",
            "count": 7,
            "exclusive": false,
            "exhaustive": false,
            "id": "genebench-pro-terminal-proteomics-biomarkers",
            "label": "Terminal subdomain — Proteomics and biomarkers",
            "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
            "reporting_status": "reported"
          },
          {
            "basis": "problems in the terminal-subdomain atlas",
            "count": 11,
            "exclusive": false,
            "exhaustive": false,
            "id": "genebench-pro-terminal-clinical-variant-penetrance",
            "label": "Terminal subdomain — Clinical variant interpretation & penetrance",
            "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
            "reporting_status": "reported"
          },
          {
            "basis": "problems in the terminal-subdomain atlas",
            "count": 8,
            "exclusive": false,
            "exhaustive": false,
            "id": "genebench-pro-terminal-pharmacogenomics-treatment-response",
            "label": "Terminal subdomain — Pharmacogenomics and treatment response",
            "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
            "reporting_status": "reported"
          },
          {
            "basis": "problems in the terminal-subdomain atlas",
            "count": 7,
            "exclusive": false,
            "exhaustive": false,
            "id": "genebench-pro-terminal-prenatal-reproductive-risk",
            "label": "Terminal subdomain — Prenatal, reproductive, and clinical-risk genetics",
            "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
            "reporting_status": "reported"
          },
          {
            "basis": "problems in the terminal-subdomain atlas",
            "count": 10,
            "exclusive": false,
            "exhaustive": false,
            "id": "genebench-pro-terminal-cancer-somatic-liquid-biopsy",
            "label": "Terminal subdomain — Cancer somatic genomics and liquid biopsy",
            "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
            "reporting_status": "reported"
          },
          {
            "basis": "problems in the terminal-subdomain atlas",
            "count": 3,
            "exclusive": false,
            "exhaustive": false,
            "id": "genebench-pro-terminal-microbial-metagenomic-genomics",
            "label": "Terminal subdomain — Microbial and metagenomic genomics",
            "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
            "reporting_status": "reported"
          },
          {
            "basis": "problems in the terminal-subdomain atlas",
            "count": 2,
            "exclusive": false,
            "exhaustive": false,
            "id": "genebench-pro-terminal-forensic-genetics-terminal",
            "label": "Terminal subdomain — Forensic genetics",
            "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
            "reporting_status": "reported"
          },
          {
            "basis": "problems",
            "count": 82,
            "exclusive": false,
            "exhaustive": false,
            "id": "genebench-pro-externally-reviewed",
            "label": "Externally reviewed problems",
            "notes": "Orthogonal review-status stratum; all ten public problems are included.",
            "reporting_status": "reported"
          },
          {
            "basis": "problems",
            "count": 47,
            "exclusive": false,
            "exhaustive": false,
            "id": "genebench-pro-not-externally-reviewed",
            "label": "Not externally reviewed",
            "notes": "Orthogonal review-status stratum.",
            "reporting_status": "reported"
          }
        ],
        "total": 129
      },
      "task_formats": [
        "isolated-workspace scientific analysis",
        "exact JSON response"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Audited against the formal paper, official OpenAI launch/case-study pages, and commit-pinned ten-problem public package.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-06-30",
          "evidence_ids": [
            "genebench-pro-paper-evidence",
            "genebench-pro-public-package-evidence"
          ],
          "formal_tracks": [],
          "id": "genebench-pro-paper-v1",
          "label": "paper-v1",
          "notes": "Initial formal 129-problem suite reported in bioRxiv v1 and the official OpenAI PDF. The public Hugging Face package is a ten-problem reproducibility subset, not a replacement full-suite version.",
          "release_date": "2026-06-30",
          "status": "current",
          "task_counts": {
            "basis": "self-contained synthetic scientific-analysis problems (called evaluations in the paper abstract)",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "problems",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "genebench-pro-public-release",
                "label": "Public release subset",
                "notes": "Ten externally reviewed case-study packages are public with prompts, staged data, ground truth, grader configuration, and reports.",
                "reporting_status": "reported"
              },
              {
                "basis": "held-out problems",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "genebench-pro-artificial-analysis",
                "label": "Artificial Analysis reporting subset",
                "notes": "Disjoint from the public release. The formal paper says this locked reporting subset was provided to Artificial Analysis; the launch page still describes delivery in future tense. The tasks are not public.",
                "reporting_status": "reported"
              },
              {
                "basis": "held-out problems",
                "count": 69,
                "exclusive": true,
                "exhaustive": true,
                "id": "genebench-pro-internal-holdout",
                "label": "Internal holdout",
                "notes": "Remainder after the disjoint 10-problem public and 50-problem Artificial Analysis subsets; not publicly released.",
                "reporting_status": "reported"
              },
              {
                "basis": "problems in the primary-domain atlas",
                "count": 17,
                "exclusive": false,
                "exhaustive": false,
                "id": "genebench-pro-primary-statistical-genetics",
                "label": "Primary domain — Statistical genetics",
                "notes": "One member of the 10-domain partition; marked non-exhaustive here because the registry also stores independent release and review partitions.",
                "reporting_status": "reported"
              },
              {
                "basis": "problems in the primary-domain atlas",
                "count": 21,
                "exclusive": false,
                "exhaustive": false,
                "id": "genebench-pro-primary-population-genetics",
                "label": "Primary domain — Population genetics",
                "notes": "One member of the 10-domain partition; marked non-exhaustive here because the registry also stores independent release and review partitions.",
                "reporting_status": "reported"
              },
              {
                "basis": "problems in the primary-domain atlas",
                "count": 17,
                "exclusive": false,
                "exhaustive": false,
                "id": "genebench-pro-primary-quantitative-genetics",
                "label": "Primary domain — Quantitative genetics",
                "notes": "One member of the 10-domain partition; marked non-exhaustive here because the registry also stores independent release and review partitions.",
                "reporting_status": "reported"
              },
              {
                "basis": "problems in the primary-domain atlas",
                "count": 17,
                "exclusive": false,
                "exhaustive": false,
                "id": "genebench-pro-primary-regulatory-omics",
                "label": "Primary domain — Regulatory omics",
                "notes": "One member of the 10-domain partition; marked non-exhaustive here because the registry also stores independent release and review partitions.",
                "reporting_status": "reported"
              },
              {
                "basis": "problems in the primary-domain atlas",
                "count": 9,
                "exclusive": false,
                "exhaustive": false,
                "id": "genebench-pro-primary-functional-genomics",
                "label": "Primary domain — Functional genomics",
                "notes": "One member of the 10-domain partition; marked non-exhaustive here because the registry also stores independent release and review partitions.",
                "reporting_status": "reported"
              },
              {
                "basis": "problems in the primary-domain atlas",
                "count": 7,
                "exclusive": false,
                "exhaustive": false,
                "id": "genebench-pro-primary-proteomics",
                "label": "Primary domain — Proteomics",
                "notes": "One member of the 10-domain partition; marked non-exhaustive here because the registry also stores independent release and review partitions.",
                "reporting_status": "reported"
              },
              {
                "basis": "problems in the primary-domain atlas",
                "count": 26,
                "exclusive": false,
                "exhaustive": false,
                "id": "genebench-pro-primary-clinical-pgx-diagnostics",
                "label": "Primary domain — Clinical, PGx & diagnostics",
                "notes": "One member of the 10-domain partition; marked non-exhaustive here because the registry also stores independent release and review partitions.",
                "reporting_status": "reported"
              },
              {
                "basis": "problems in the primary-domain atlas",
                "count": 10,
                "exclusive": false,
                "exhaustive": false,
                "id": "genebench-pro-primary-cancer-genomics",
                "label": "Primary domain — Cancer genomics",
                "notes": "One member of the 10-domain partition; marked non-exhaustive here because the registry also stores independent release and review partitions.",
                "reporting_status": "reported"
              },
              {
                "basis": "problems in the primary-domain atlas",
                "count": 3,
                "exclusive": false,
                "exhaustive": false,
                "id": "genebench-pro-primary-microbial-genomics",
                "label": "Primary domain — Microbial genomics",
                "notes": "One member of the 10-domain partition; marked non-exhaustive here because the registry also stores independent release and review partitions.",
                "reporting_status": "reported"
              },
              {
                "basis": "problems in the primary-domain atlas",
                "count": 2,
                "exclusive": false,
                "exhaustive": false,
                "id": "genebench-pro-primary-forensic-genetics",
                "label": "Primary domain — Forensic genetics",
                "notes": "One member of the 10-domain partition; marked non-exhaustive here because the registry also stores independent release and review partitions.",
                "reporting_status": "reported"
              },
              {
                "basis": "problems in the terminal-subdomain atlas",
                "count": 6,
                "exclusive": false,
                "exhaustive": false,
                "id": "genebench-pro-terminal-association-correction",
                "label": "Terminal subdomain — Association & correction",
                "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
                "reporting_status": "reported"
              },
              {
                "basis": "problems in the terminal-subdomain atlas",
                "count": 6,
                "exclusive": false,
                "exhaustive": false,
                "id": "genebench-pro-terminal-causal-mapping",
                "label": "Terminal subdomain — Causal mapping",
                "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
                "reporting_status": "reported"
              },
              {
                "basis": "problems in the terminal-subdomain atlas",
                "count": 2,
                "exclusive": false,
                "exhaustive": false,
                "id": "genebench-pro-terminal-heritability-architecture",
                "label": "Terminal subdomain — Heritability and architecture",
                "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
                "reporting_status": "reported"
              },
              {
                "basis": "problems in the terminal-subdomain atlas",
                "count": 3,
                "exclusive": false,
                "exhaustive": false,
                "id": "genebench-pro-terminal-pedigree-ibd-phasing",
                "label": "Terminal subdomain — Pedigree, IBD, and phasing",
                "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
                "reporting_status": "reported"
              },
              {
                "basis": "problems in the terminal-subdomain atlas",
                "count": 7,
                "exclusive": false,
                "exhaustive": false,
                "id": "genebench-pro-terminal-selection-mutation",
                "label": "Terminal subdomain — Selection & mutation",
                "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
                "reporting_status": "reported"
              },
              {
                "basis": "problems in the terminal-subdomain atlas",
                "count": 6,
                "exclusive": false,
                "exhaustive": false,
                "id": "genebench-pro-terminal-admixture-adna",
                "label": "Terminal subdomain — Admixture & ancient DNA",
                "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
                "reporting_status": "reported"
              },
              {
                "basis": "problems in the terminal-subdomain atlas",
                "count": 8,
                "exclusive": false,
                "exhaustive": false,
                "id": "genebench-pro-terminal-history-genealogies",
                "label": "Terminal subdomain — History & genealogies",
                "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
                "reporting_status": "reported"
              },
              {
                "basis": "problems in the terminal-subdomain atlas",
                "count": 6,
                "exclusive": false,
                "exhaustive": false,
                "id": "genebench-pro-terminal-trait-architecture-variance",
                "label": "Terminal subdomain — Trait architecture and variance",
                "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
                "reporting_status": "reported"
              },
              {
                "basis": "problems in the terminal-subdomain atlas",
                "count": 6,
                "exclusive": false,
                "exhaustive": false,
                "id": "genebench-pro-terminal-family-social-transmission",
                "label": "Terminal subdomain — Family, social, and transmission effects",
                "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
                "reporting_status": "reported"
              },
              {
                "basis": "problems in the terminal-subdomain atlas",
                "count": 5,
                "exclusive": false,
                "exhaustive": false,
                "id": "genebench-pro-terminal-polygenic-prediction-selection",
                "label": "Terminal subdomain — Polygenic prediction and genomic selection",
                "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
                "reporting_status": "reported"
              },
              {
                "basis": "problems in the terminal-subdomain atlas",
                "count": 8,
                "exclusive": false,
                "exhaustive": false,
                "id": "genebench-pro-terminal-regulatory-qtls-ase",
                "label": "Terminal subdomain — Regulatory QTLs & ASE",
                "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
                "reporting_status": "reported"
              },
              {
                "basis": "problems in the terminal-subdomain atlas",
                "count": 5,
                "exclusive": false,
                "exhaustive": false,
                "id": "genebench-pro-terminal-transcriptome-structure",
                "label": "Terminal subdomain — Transcriptome structure",
                "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
                "reporting_status": "reported"
              },
              {
                "basis": "problems in the terminal-subdomain atlas",
                "count": 4,
                "exclusive": false,
                "exhaustive": false,
                "id": "genebench-pro-terminal-spatial-chromatin-context",
                "label": "Terminal subdomain — Spatial and chromatin context",
                "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
                "reporting_status": "reported"
              },
              {
                "basis": "problems in the terminal-subdomain atlas",
                "count": 9,
                "exclusive": false,
                "exhaustive": false,
                "id": "genebench-pro-terminal-functional-genomics-terminal",
                "label": "Terminal subdomain — Functional genomics",
                "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
                "reporting_status": "reported"
              },
              {
                "basis": "problems in the terminal-subdomain atlas",
                "count": 7,
                "exclusive": false,
                "exhaustive": false,
                "id": "genebench-pro-terminal-proteomics-biomarkers",
                "label": "Terminal subdomain — Proteomics and biomarkers",
                "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
                "reporting_status": "reported"
              },
              {
                "basis": "problems in the terminal-subdomain atlas",
                "count": 11,
                "exclusive": false,
                "exhaustive": false,
                "id": "genebench-pro-terminal-clinical-variant-penetrance",
                "label": "Terminal subdomain — Clinical variant interpretation & penetrance",
                "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
                "reporting_status": "reported"
              },
              {
                "basis": "problems in the terminal-subdomain atlas",
                "count": 8,
                "exclusive": false,
                "exhaustive": false,
                "id": "genebench-pro-terminal-pharmacogenomics-treatment-response",
                "label": "Terminal subdomain — Pharmacogenomics and treatment response",
                "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
                "reporting_status": "reported"
              },
              {
                "basis": "problems in the terminal-subdomain atlas",
                "count": 7,
                "exclusive": false,
                "exhaustive": false,
                "id": "genebench-pro-terminal-prenatal-reproductive-risk",
                "label": "Terminal subdomain — Prenatal, reproductive, and clinical-risk genetics",
                "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
                "reporting_status": "reported"
              },
              {
                "basis": "problems in the terminal-subdomain atlas",
                "count": 10,
                "exclusive": false,
                "exhaustive": false,
                "id": "genebench-pro-terminal-cancer-somatic-liquid-biopsy",
                "label": "Terminal subdomain — Cancer somatic genomics and liquid biopsy",
                "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
                "reporting_status": "reported"
              },
              {
                "basis": "problems in the terminal-subdomain atlas",
                "count": 3,
                "exclusive": false,
                "exhaustive": false,
                "id": "genebench-pro-terminal-microbial-metagenomic-genomics",
                "label": "Terminal subdomain — Microbial and metagenomic genomics",
                "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
                "reporting_status": "reported"
              },
              {
                "basis": "problems in the terminal-subdomain atlas",
                "count": 2,
                "exclusive": false,
                "exhaustive": false,
                "id": "genebench-pro-terminal-forensic-genetics-terminal",
                "label": "Terminal subdomain — Forensic genetics",
                "notes": "One member of the 21-terminal-subdomain partition; stored alongside independent release and review partitions.",
                "reporting_status": "reported"
              },
              {
                "basis": "problems",
                "count": 82,
                "exclusive": false,
                "exhaustive": false,
                "id": "genebench-pro-externally-reviewed",
                "label": "Externally reviewed problems",
                "notes": "Orthogonal review-status stratum; all ten public problems are included.",
                "reporting_status": "reported"
              },
              {
                "basis": "problems",
                "count": 47,
                "exclusive": false,
                "exhaustive": false,
                "id": "genebench-pro-not-externally-reviewed",
                "label": "Not externally reviewed",
                "notes": "Orthogonal review-status stratum.",
                "reporting_status": "reported"
              }
            ],
            "total": 129
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official package publishes dataset construction notebooks and TensorFlow/PyTorch baseline workflows.",
        "biosafety_notes": "The benchmark concerns public genomic annotations; this registry mirrors no sequences.",
        "grader": "Deterministic classification accuracy and F1 calculations on held-out test sets.",
        "level": "fully-open",
        "license": "Apache-2.0 for package code; reference genomes and adapted source datasets retain source-specific terms.",
        "tasks": "Versioned coordinates, train/test splits, cloud-cached sequences, and Python/Hugging Face loaders are public."
      },
      "aliases": [
        "Genomic Benchmarks collection"
      ],
      "audit": {
        "audited_date": "2026-07-22",
        "notes": "Nine paper datasets, their categories, package version, and current metadata versions were checked separately.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction",
        "classification"
      ],
      "coverage_notes": [
        {
          "count": 9,
          "coverage": "explicitly-in-scope",
          "notes": "All nine formal entries are genomic DNA sequence-classification datasets.",
          "reporting_status": "reported",
          "tag": "genomics"
        },
        {
          "count": 1,
          "coverage": "explicitly-in-scope",
          "notes": "The human open-chromatin-region dataset directly tests chromatin accessibility classification.",
          "reporting_status": "reported",
          "tag": "epigenomics"
        }
      ],
      "domains": [
        "genomics",
        "epigenomics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "genomic-benchmarks-paper"
      ],
      "evaluation_run_ids": [
        "genomic-benchmarks-creator-full"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "genomic-benchmarks-paper-definition-evidence",
          "locator": {
            "note": "Lists nine datasets, sequence counts, classes, categories, train/test design, and creator CNN metrics.",
            "type": "table",
            "value": "Construction and content, Overview of Datasets, Tables 1-2"
          },
          "source_id": "genomic-benchmarks-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/resources",
            "/versions/0/release_date",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0",
            "/scientific_task_classification/entries/1",
            "/scientific_task_classification/entries/2",
            "/scientific_task_classification/entries/3"
          ]
        },
        {
          "accessed_date": "2026-07-22",
          "id": "genomic-benchmarks-repository-evidence",
          "locator": {
            "note": "Confirms package version, nine dataset directories, version metadata, public loaders, licenses, and baseline metrics.",
            "type": "repository-path",
            "value": "README.md; setup.py; datasets/*/metadata.yaml; experiments/README.md at 605d8539830e16c85abe7826990958303ffc5e1c"
          },
          "source_id": "genomic-benchmarks-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/implementations",
            "/versions/0/as_of",
            "/versions/0/formal_tracks"
          ]
        }
      ],
      "field_status": [],
      "id": "genomic-benchmarks",
      "implementations": [
        {
          "commit": "605d8539830e16c85abe7826990958303ffc5e1c",
          "framework": "genomic-benchmarks Python package",
          "notes": "Package version 1.0.0 with dataset loaders and creator CNN baselines.",
          "status": "official",
          "url": "https://github.com/ML-Bioinfo-CEITEC/genomic_benchmarks/tree/605d8539830e16c85abe7826990958303ffc5e1c"
        }
      ],
      "kind": "suite",
      "latest_version": "package-1.0.0-snapshot",
      "modalities": [
        "dna-rna-sequence"
      ],
      "name": "Genomic Benchmarks",
      "organizations": [
        "Masaryk University",
        "CEITEC Masaryk University"
      ],
      "parent_id": null,
      "release_date": "2023-05-01",
      "resources": [
        {
          "access_notes": "Peer-reviewed creator paper in BMC Genomic Data.",
          "id": "genomic-benchmarks-paper-resource",
          "last_checked": "2026-07-22",
          "license": "CC BY 4.0",
          "pin": null,
          "type": "paper",
          "url": "https://doi.org/10.1186/s12863-023-01123-8"
        },
        {
          "access_notes": "Official package and dataset-construction repository; all nine metadata files report dataset version 0 at the pinned commit.",
          "id": "genomic-benchmarks-repository-resource",
          "last_checked": "2026-07-22",
          "license": "Apache-2.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/ML-Bioinfo-CEITEC/genomic_benchmarks/tree/605d8539830e16c85abe7826990958303ffc5e1c",
            "value": "605d8539830e16c85abe7826990958303ffc5e1c"
          },
          "type": "repository",
          "url": "https://github.com/ML-Bioinfo-CEITEC/genomic_benchmarks"
        }
      ],
      "scientific_task_classification": {
        "as_of": "2026-07-22",
        "benchmark_version": "package-1.0.0-snapshot",
        "entries": [
          {
            "confidence": "high",
            "count": 4,
            "count_basis": "Benchmark dataset classification tasks.",
            "count_ref": null,
            "count_unit": "tasks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "genomic-benchmarks-paper-definition-evidence"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "Mouse enhancer, Drosophila enhancer, human enhancer Cohn, and human enhancer Ensembl datasets.",
            "reporting_status": "reported",
            "task_type_id": "enhancer-activity-prediction"
          },
          {
            "confidence": "high",
            "count": 1,
            "count_basis": "Benchmark dataset classification tasks.",
            "count_ref": null,
            "count_unit": "tasks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "genomic-benchmarks-paper-definition-evidence"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "Human non-TATA promoter classification.",
            "reporting_status": "reported",
            "task_type_id": "promoter-detection"
          },
          {
            "confidence": "high",
            "count": 1,
            "count_basis": "Benchmark dataset classification tasks.",
            "count_ref": null,
            "count_unit": "tasks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "genomic-benchmarks-paper-definition-evidence"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "Human open-chromatin region classification.",
            "reporting_status": "reported",
            "task_type_id": "epigenetic-mark-prediction"
          },
          {
            "confidence": "high",
            "count": 3,
            "count_basis": "Benchmark dataset classification tasks.",
            "count_ref": null,
            "count_unit": "tasks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "genomic-benchmarks-paper-definition-evidence"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "Two demo coding-versus-intergenic datasets and the human regulatory-region multiclass dataset.",
            "reporting_status": "reported",
            "task_type_id": "dna-sequence-analysis"
          }
        ],
        "notes": "Each of the nine version-0 dataset directories is assigned once using the creator paper's Table 1 labels.",
        "status": "complete"
      },
      "summary": "A versioned collection of nine DNA sequence-classification datasets covering regulatory elements, promoters, enhancers, open chromatin, species, and coding-context discrimination.",
      "task_counts": {
        "basis": "benchmark dataset classification tasks",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "benchmark dataset classification tasks",
            "count": 6,
            "exclusive": true,
            "exhaustive": true,
            "id": "genomic-benchmarks-core-regulatory",
            "label": "Core regulatory-element datasets",
            "notes": "Drosophila enhancer plus five human regulatory datasets.",
            "reporting_status": "reported"
          },
          {
            "basis": "benchmark dataset classification tasks",
            "count": 2,
            "exclusive": true,
            "exhaustive": true,
            "id": "genomic-benchmarks-demo",
            "label": "Demonstration datasets",
            "notes": "Coding-versus-intergenic and human-versus-worm demos.",
            "reporting_status": "reported"
          },
          {
            "basis": "benchmark dataset classification tasks",
            "count": 1,
            "exclusive": true,
            "exhaustive": true,
            "id": "genomic-benchmarks-dummy",
            "label": "Dummy dataset",
            "notes": "Small mouse-enhancer prototyping dataset.",
            "reporting_status": "reported"
          }
        ],
        "total": 9
      },
      "task_formats": [
        "binary genomic sequence classification",
        "multiclass genomic sequence classification"
      ],
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Paper Table 1 and the pinned repository enumerate the same nine datasets.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-07-22",
          "evidence_ids": [
            "genomic-benchmarks-paper-definition-evidence",
            "genomic-benchmarks-repository-evidence"
          ],
          "formal_tracks": [],
          "id": "genomic-benchmarks-package-1-0-0-snapshot",
          "label": "package-1.0.0-snapshot",
          "notes": "As-of snapshot of the nine datasets enumerated in the paper and pinned repository; future repository additions require a new snapshot.",
          "release_date": "2023-05-01",
          "status": "rolling",
          "task_counts": {
            "basis": "benchmark dataset classification tasks",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "benchmark dataset classification tasks",
                "count": 6,
                "exclusive": true,
                "exhaustive": true,
                "id": "genomic-benchmarks-core-regulatory",
                "label": "Core regulatory-element datasets",
                "notes": "Drosophila enhancer plus five human regulatory datasets.",
                "reporting_status": "reported"
              },
              {
                "basis": "benchmark dataset classification tasks",
                "count": 2,
                "exclusive": true,
                "exhaustive": true,
                "id": "genomic-benchmarks-demo",
                "label": "Demonstration datasets",
                "notes": "Coding-versus-intergenic and human-versus-worm demos.",
                "reporting_status": "reported"
              },
              {
                "basis": "benchmark dataset classification tasks",
                "count": 1,
                "exclusive": true,
                "exhaustive": true,
                "id": "genomic-benchmarks-dummy",
                "label": "Dummy dataset",
                "notes": "Small mouse-enhancer prototyping dataset.",
                "reporting_status": "reported"
              }
            ],
            "total": 9
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official package provides scorers; a companion creator repository provides baselines and a Docker environment.",
        "biosafety_notes": "The benchmark optimizes drug-like small molecules; this registry mirrors no molecular datasets or generated compounds.",
        "grader": "Deterministic validity, uniqueness, novelty, KL, FCD, and goal-directed scoring functions with scores normalized to higher-is-better.",
        "level": "fully-open",
        "license": "MIT for GuacaMol code; the standardized molecular data derive from ChEMBL and retain applicable ChEMBL terms.",
        "tasks": "Both benchmark suites, standardized ChEMBL-derived train/validation/test files, and the holdout set are public."
      },
      "aliases": [
        "GuacaMol generative chemistry benchmark"
      ],
      "audit": {
        "audited_date": "2026-07-22",
        "notes": "Suite counts were reproduced directly from the commit-pinned benchmark_suites.py; generated sample counts are kept separate.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "generation",
        "optimization"
      ],
      "coverage_notes": [
        {
          "count": 25,
          "coverage": "explicitly-in-scope",
          "notes": "Counts formal evaluation problems, not generated molecules or scoring-function calls.",
          "reporting_status": "reported",
          "tag": "medchem"
        }
      ],
      "domains": [
        "medchem"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "guacamol-paper"
      ],
      "evaluation_run_ids": [
        "guacamol-creator-full"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "guacamol-paper-definition-evidence",
          "locator": {
            "note": "Defines distribution learning, goal-directed generation, standardized ChEMBL data, baseline evaluation, and score semantics.",
            "type": "section",
            "value": "Sections 2-4 and Tables 1-2"
          },
          "source_id": "guacamol-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/resources",
            "/versions/0/release_date",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-22",
          "id": "guacamol-repository-evidence",
          "locator": {
            "note": "Confirms package version, five distribution benchmarks, twenty goal-directed v2 problems, public data hashes, and MIT license.",
            "type": "repository-path",
            "value": "README.md; guacamol/__init__.py; guacamol/benchmark_suites.py at 60ebe1f6a396f16e08b834dce448e9343d259feb"
          },
          "source_id": "guacamol-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/subsets",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/implementations",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/subsets",
            "/versions/0/formal_tracks",
            "/scientific_task_classification/entries/0"
          ]
        }
      ],
      "field_status": [],
      "id": "guacamol",
      "implementations": [
        {
          "commit": "60ebe1f6a396f16e08b834dce448e9343d259feb",
          "framework": "GuacaMol 0.5.5",
          "notes": "Reproducible distribution-learning and goal-directed suite runners and scoring functions.",
          "status": "official",
          "url": "https://github.com/BenevolentAI/guacamol/tree/60ebe1f6a396f16e08b834dce448e9343d259feb"
        },
        {
          "commit": "ae43219c89d5db134028336243f508606d81995e",
          "framework": "GuacaMol baselines",
          "notes": "Companion creator baselines and Dockerfile.",
          "status": "official",
          "url": "https://github.com/BenevolentAI/guacamol_baselines/tree/ae43219c89d5db134028336243f508606d81995e"
        }
      ],
      "kind": "suite",
      "latest_version": "suite-v2",
      "modalities": [
        "small-molecule-structure"
      ],
      "name": "GuacaMol",
      "organizations": [
        "BenevolentAI"
      ],
      "parent_id": null,
      "release_date": "2018-11-23",
      "resources": [
        {
          "access_notes": "Peer-reviewed creator paper in Journal of Chemical Information and Modeling.",
          "id": "guacamol-paper-resource",
          "last_checked": "2026-07-22",
          "license": null,
          "pin": null,
          "type": "paper",
          "url": "https://doi.org/10.1021/acs.jcim.8b00839"
        },
        {
          "access_notes": "Official benchmark package version 0.5.5 with v1/v2 suite definitions.",
          "id": "guacamol-repository-resource",
          "last_checked": "2026-07-22",
          "license": "MIT",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/BenevolentAI/guacamol/tree/60ebe1f6a396f16e08b834dce448e9343d259feb",
            "value": "60ebe1f6a396f16e08b834dce448e9343d259feb"
          },
          "type": "repository",
          "url": "https://github.com/BenevolentAI/guacamol"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "suite-v2",
        "entries": [
          {
            "confidence": "high",
            "count": 25,
            "count_basis": "Formal benchmark problems.",
            "count_ref": "/task_counts/total",
            "count_unit": "problems",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "guacamol-paper-definition-evidence",
              "guacamol-repository-evidence"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "Includes both distribution-learning assessment and goal-directed molecular optimization problems.",
            "reporting_status": "reported",
            "task_type_id": "small-molecule-generation"
          }
        ],
        "notes": "Version 2 of the official benchmark suite defines five distribution-learning and twenty goal-directed problems.",
        "status": "complete"
      },
      "summary": "A reproducible benchmark for de novo molecular design with five distribution-learning tests and twenty goal-directed generation problems in the current v2 suite.",
      "task_counts": {
        "basis": "formal benchmark problems",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "formal benchmark problems",
            "count": 5,
            "exclusive": true,
            "exhaustive": true,
            "id": "guacamol-distribution-learning",
            "label": "Distribution-learning benchmarks",
            "notes": "Validity, uniqueness, novelty, KL divergence, and Fréchet ChemNet Distance.",
            "reporting_status": "reported"
          },
          {
            "basis": "formal benchmark problems",
            "count": 20,
            "exclusive": true,
            "exhaustive": true,
            "id": "guacamol-goal-directed-v2",
            "label": "Goal-directed v2 benchmarks",
            "notes": "Rediscovery, similarity, isomer, median-molecule, and multi-property optimization problems.",
            "reporting_status": "reported"
          }
        ],
        "total": 25
      },
      "task_formats": [
        "unconditional molecular generation",
        "goal-directed molecular optimization"
      ],
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Creator paper and v0.5.5 implementation define the two benchmark modes and their scorers.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": null,
          "evidence_ids": [
            "guacamol-paper-definition-evidence",
            "guacamol-repository-evidence"
          ],
          "formal_tracks": [],
          "id": "guacamol-suite-v2",
          "label": "suite-v2",
          "notes": "Distribution-learning v1/v2 definitions are identical; the current goal-directed v2 list contains twenty problems.",
          "release_date": "2018-11-23",
          "status": "current",
          "task_counts": {
            "basis": "formal benchmark problems",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "formal benchmark problems",
                "count": 5,
                "exclusive": true,
                "exhaustive": true,
                "id": "guacamol-distribution-learning",
                "label": "Distribution-learning benchmarks",
                "notes": "Validity, uniqueness, novelty, KL divergence, and Fréchet ChemNet Distance.",
                "reporting_status": "reported"
              },
              {
                "basis": "formal benchmark problems",
                "count": 20,
                "exclusive": true,
                "exhaustive": true,
                "id": "guacamol-goal-directed-v2",
                "label": "Goal-directed v2 benchmarks",
                "notes": "Rediscovery, similarity, isomer, median-molecule, and multi-property optimization problems.",
                "reporting_status": "reported"
              }
            ],
            "total": 25
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "Public questions, figures, tables, open-answer variants, split manifests, and the Python harness are released.",
        "biosafety_notes": "The suite includes molecular-cloning and viral-PPI questions. BioBench Atlas stores only metadata and aggregate results and does not mirror questions or private content.",
        "grader": "The official harness scores accuracy, precision/selective accuracy, coverage, and total n from exact single-choice outputs.",
        "level": "partially-open",
        "license": "CC-BY-SA-4.0",
        "tasks": "1,967 questions are public and 490 question contents are withheld; all 2,457 IDs and split membership are versioned."
      },
      "aliases": [
        "Language Agent Biology Benchmark"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Counts, access, protocol, and result tables are audited. The README claim of 30 narrower subtasks conflicts with 31 versioned split files and 31 creator-result rows.",
        "status": "audited-with-caveats",
        "unresolved_fields": 2
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "knowledge",
        "evidence-synthesis",
        "retrieval",
        "prediction",
        "classification",
        "design",
        "data-analysis",
        "tool-use",
        "experiment-planning",
        "troubleshooting",
        "scientific-reasoning"
      ],
      "coverage_notes": [
        {
          "count": null,
          "coverage": "explicitly-in-scope",
          "notes": "SeqQA and two ClinVar DbQA tasks require DNA/protein-sequence reasoning, but the sources do not publish a protein-only question count.",
          "reporting_status": "not_reported",
          "tag": "protein-sequence"
        },
        {
          "count": 50,
          "coverage": "explicitly-in-scope",
          "notes": "The Viral PPI formal task has 50 questions about predicted viral–human protein interactions.",
          "reporting_status": "reported",
          "tag": "protein-protein-binding"
        },
        {
          "count": 0,
          "coverage": "not-in-scope",
          "notes": "Primer and cloning design are in scope; protein sequence design is not a released LAB-Bench task.",
          "reporting_status": "reported",
          "tag": "protein-design"
        }
      ],
      "domains": [
        "life-science",
        "molecular-cell-biology",
        "assay-screening",
        "bioinformatics",
        "protein-sequence",
        "protein-protein-binding",
        "genomics",
        "transcriptomics",
        "epigenomics",
        "clinical-translational",
        "virology"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-evidence-paper",
          "locator": {
            "note": "Definition, full count, category/subtask rows, public/private policy, task descriptions, creator protocol, metrics, models, and results.",
            "type": "section",
            "value": "Abstract; Sections 1–2; Table 1; Appendix B–E; Tables 2–6"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/organizations",
            "/release_date",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/task_counts/subsets/10/count",
            "/coverage_notes",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/1",
            "/scientific_task_classification/entries/2",
            "/scientific_task_classification/entries/3",
            "/scientific_task_classification/entries/4"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-evidence-repository",
          "locator": {
            "note": "31 files sum to public/private/total 1,967/490/2,457; README states 8 categories and 30 narrower subtasks.",
            "type": "repository-path",
            "value": "README.md; CHANGELOG.md; LICENSE; 31 *-splits.json files; labbench/evaluator.py at 998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "source_id": "lab-bench-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/task_counts/subsets/10/count",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/implementations",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets",
            "/versions/1/task_counts/subsets/10/count",
            "/scientific_task_classification/entries/0"
          ]
        }
      ],
      "field_status": [
        {
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-evidence-paper",
            "lab-bench-evidence-repository"
          ],
          "path": "/task_counts/subsets/10/count",
          "reason": "The official README says 30 narrower subtasks, while the pinned repository contains 31 versioned split files and the paper reports 31 task/result rows.",
          "status": "conflicted"
        },
        {
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-evidence-repository"
          ],
          "path": "/versions/1/task_counts/subsets/10/count",
          "reason": "The current snapshot preserves the README claim of 30 alongside the reproducible count of 31 versioned split files.",
          "status": "conflicted"
        }
      ],
      "id": "lab-bench",
      "implementations": [
        {
          "commit": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838",
          "framework": "labbench Python evaluator",
          "notes": "Python 3.10+ harness with public task loaders, exact-choice scoring, and accuracy/precision/coverage metrics.",
          "status": "official",
          "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/labbench"
        }
      ],
      "kind": "suite",
      "latest_version": "repository-998a8e0",
      "modalities": [
        "text",
        "paper",
        "table",
        "figure",
        "image",
        "dna-rna-sequence",
        "protein-sequence",
        "database",
        "web"
      ],
      "name": "LAB-Bench",
      "organizations": [
        "FutureHouse"
      ],
      "parent_id": null,
      "release_date": "2024-07-14",
      "resources": [
        {
          "access_notes": "Creator preprint v3; the official HTML snapshot used for the audit has SHA256 4a29823b3e160541213840523b8f4ff69fdcadf4314245ab09b7e7cb1ea62e44.",
          "id": "lab-bench-paper-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "version",
            "url": "https://arxiv.org/html/2407.10362v3",
            "value": "v3"
          },
          "type": "paper",
          "url": "https://arxiv.org/abs/2407.10362v3"
        },
        {
          "access_notes": "Official harness, public data, 31 split manifests, and private IDs.",
          "id": "lab-bench-repository-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-SA-4.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838",
            "value": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "type": "repository",
          "url": "https://github.com/Future-House/LAB-Bench"
        },
        {
          "access_notes": "Official public dataset mirror; ungated at verification time.",
          "id": "lab-bench-dataset-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-SA-4.0",
          "pin": {
            "kind": "commit",
            "url": "https://huggingface.co/datasets/futurehouse/lab-bench/tree/5c77cec648430f30611808808861eb86f81d5eaa",
            "value": "5c77cec648430f30611808808861eb86f81d5eaa"
          },
          "type": "dataset",
          "url": "https://huggingface.co/datasets/futurehouse/lab-bench"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "repository-998a8e0",
        "entries": [
          {
            "confidence": "high",
            "count": 650,
            "count_basis": "Questions across the ten formal DbQA child tasks.",
            "count_ref": null,
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-evidence-repository"
            ],
            "mapping_method": "official-track",
            "notes": "Count is specific to DbQA and is not added to other task claims.",
            "reporting_status": "reported",
            "task_type_id": "scientific-database-retrieval"
          },
          {
            "confidence": "high",
            "count": null,
            "count_basis": "FigQA, LitQA2, SuppQA, and TableQA questions.",
            "count_ref": null,
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "The root record does not publish this cross-track subtotal.",
            "reporting_status": "not_reported",
            "task_type_id": "scientific-evidence-interpretation"
          },
          {
            "confidence": "high",
            "count": 135,
            "count_basis": "ProtocolQA questions across public and private splits.",
            "count_ref": null,
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "CloningScenarios is shown separately and is not included in this count.",
            "reporting_status": "reported",
            "task_type_id": "experiment-protocol-planning"
          },
          {
            "confidence": "high",
            "count": 50,
            "count_basis": "Viral PPI formal-task questions.",
            "count_ref": "/coverage_notes/1/count",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Viral-human PPI database-retrieval questions.",
            "reporting_status": "reported",
            "task_type_id": "protein-protein-interaction-prediction"
          },
          {
            "confidence": "high",
            "count": 0,
            "count_basis": "Released LAB-Bench questions.",
            "count_ref": "/coverage_notes/2/count",
            "count_unit": "questions",
            "coverage": "not-in-scope",
            "evidence_ids": [
              "lab-bench-evidence-paper"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "Primer and cloning design are not relabeled as protein sequence design.",
            "reporting_status": "reported",
            "task_type_id": "protein-sequence-design"
          }
        ],
        "notes": "Formal child tracks support several precise mappings; the mixed suite has no exhaustive creator scientific-task taxonomy.",
        "status": "partial"
      },
      "summary": "A practical biology-research suite of 2,457 multiple-choice questions across eight broad categories and 31 versioned task files, with public and private contamination-monitoring splits.",
      "task_counts": {
        "basis": "multiple-choice questions across the complete public and private creator snapshot",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "questions",
            "count": 248,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-litqa2",
            "label": "LitQA2",
            "notes": null,
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 102,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-suppqa",
            "label": "SuppQA",
            "notes": null,
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 226,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-figqa",
            "label": "FigQA",
            "notes": null,
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 305,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-tableqa",
            "label": "TableQA",
            "notes": null,
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 650,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-dbqa",
            "label": "DbQA",
            "notes": null,
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 135,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-protocolqa",
            "label": "ProtocolQA",
            "notes": null,
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 750,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa",
            "label": "SeqQA",
            "notes": null,
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 41,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-cloning-scenarios",
            "label": "CloningScenarios",
            "notes": null,
            "reporting_status": "reported"
          },
          {
            "basis": "categories",
            "count": 8,
            "exclusive": false,
            "exhaustive": false,
            "id": "lab-bench-broad-categories",
            "label": "Broad categories",
            "notes": "Not a question count.",
            "reporting_status": "reported"
          },
          {
            "basis": "versioned split files",
            "count": 31,
            "exclusive": false,
            "exhaustive": false,
            "id": "lab-bench-formal-task-files",
            "label": "Versioned formal task files",
            "notes": "The 31 files are the six single-task categories, 10 DbQA subtasks, 15 SeqQA subtasks, and CloningScenarios.",
            "reporting_status": "reported"
          },
          {
            "basis": "subtasks claimed in README",
            "count": 30,
            "exclusive": false,
            "exhaustive": false,
            "id": "lab-bench-readme-narrower-subtasks",
            "label": "Narrower subtasks stated in README",
            "notes": "Conflicts with the 31 versioned split files and 31 creator-result rows.",
            "reporting_status": "reported"
          }
        ],
        "total": 2457
      },
      "task_formats": [
        "multiple choice"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Audited against arXiv v3, all 31 commit-pinned split manifests, the official evaluator, the current Hugging Face revision, and official Anthropic reports.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2024-07-17",
          "evidence_ids": [
            "lab-bench-evidence-paper"
          ],
          "formal_tracks": [
            "lab-bench-litqa2",
            "lab-bench-suppqa",
            "lab-bench-figqa",
            "lab-bench-tableqa",
            "lab-bench-dbqa",
            "lab-bench-protocolqa",
            "lab-bench-seqqa",
            "lab-bench-cloning-scenarios"
          ],
          "id": "lab-bench-paper-v3",
          "label": "paper-v3",
          "notes": "Complete creator-paper evaluation snapshot, including public and private splits.",
          "release_date": "2024-07-17",
          "status": "active",
          "task_counts": {
            "basis": "multiple-choice questions across the complete public and private creator snapshot",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 248,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-litqa2",
                "label": "LitQA2",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 102,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-suppqa",
                "label": "SuppQA",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 226,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-figqa",
                "label": "FigQA",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 305,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-tableqa",
                "label": "TableQA",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 650,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa",
                "label": "DbQA",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 135,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-protocolqa",
                "label": "ProtocolQA",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 750,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa",
                "label": "SeqQA",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 41,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-cloning-scenarios",
                "label": "CloningScenarios",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "categories",
                "count": 8,
                "exclusive": false,
                "exhaustive": false,
                "id": "lab-bench-broad-categories",
                "label": "Broad categories",
                "notes": "Not a question count.",
                "reporting_status": "reported"
              },
              {
                "basis": "task rows reported across Tables 2–4",
                "count": 31,
                "exclusive": false,
                "exhaustive": false,
                "id": "lab-bench-formal-task-files",
                "label": "Creator-paper task/result rows",
                "notes": "The 31 rows are the six other single-task categories, 10 DbQA subtasks, 15 SeqQA subtasks, and CloningScenarios.",
                "reporting_status": "reported"
              }
            ],
            "total": 2457
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "lab-bench-evidence-repository"
          ],
          "formal_tracks": [
            "lab-bench-litqa2",
            "lab-bench-suppqa",
            "lab-bench-figqa",
            "lab-bench-tableqa",
            "lab-bench-dbqa",
            "lab-bench-protocolqa",
            "lab-bench-seqqa",
            "lab-bench-cloning-scenarios"
          ],
          "id": "lab-bench-repository-998a8e0",
          "label": "repository-998a8e0",
          "notes": "Current official repository snapshot; includes the 2025 fixes documented in CHANGELOG.md.",
          "release_date": "2025-09-27",
          "status": "current",
          "task_counts": {
            "basis": "multiple-choice questions across the complete public and private creator snapshot",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 248,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-litqa2",
                "label": "LitQA2",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 102,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-suppqa",
                "label": "SuppQA",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 226,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-figqa",
                "label": "FigQA",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 305,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-tableqa",
                "label": "TableQA",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 650,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa",
                "label": "DbQA",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 135,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-protocolqa",
                "label": "ProtocolQA",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 750,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa",
                "label": "SeqQA",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 41,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-cloning-scenarios",
                "label": "CloningScenarios",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "categories",
                "count": 8,
                "exclusive": false,
                "exhaustive": false,
                "id": "lab-bench-broad-categories",
                "label": "Broad categories",
                "notes": "Not a question count.",
                "reporting_status": "reported"
              },
              {
                "basis": "versioned split files",
                "count": 31,
                "exclusive": false,
                "exhaustive": false,
                "id": "lab-bench-formal-task-files",
                "label": "Versioned formal task files",
                "notes": "The 31 files are the six single-task categories, 10 DbQA subtasks, 15 SeqQA subtasks, and CloningScenarios.",
                "reporting_status": "reported"
              },
              {
                "basis": "subtasks claimed in README",
                "count": 30,
                "exclusive": false,
                "exhaustive": false,
                "id": "lab-bench-readme-narrower-subtasks",
                "label": "Narrower subtasks stated in README",
                "notes": "Conflicts with the 31 versioned split files and 31 creator-result rows.",
                "reporting_status": "reported"
              }
            ],
            "total": 2457
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official repository publishes the public questions and both public/private ID manifests.",
        "biosafety_notes": "BioBench Atlas stores metadata and aggregate results only and does not mirror question or sequence content. Benchmark answers must not be treated as operational laboratory guidance.",
        "grader": "The official harness scores exact single-choice outputs and reports accuracy, precision, coverage, and n.",
        "level": "partially-open",
        "license": "CC-BY-SA-4.0",
        "tasks": "33 questions are public and 8 question contents are withheld."
      },
      "aliases": [
        "Cloning Scenarios"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": null,
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "experiment-planning",
        "design",
        "troubleshooting",
        "scientific-reasoning"
      ],
      "coverage_notes": [],
      "domains": [
        "molecular-cell-biology",
        "genomics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "anthropic-sonnet-4-5-system-card",
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-cloning-scenarios-anthropic-sonnet45-system-card",
        "lab-bench-cloning-scenarios-creator-mcq",
        "lab-bench-cloning-scenarios-creator-mcq-llama-context",
        "lab-bench-cloning-scenarios-creator-open-response"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-cloning-scenarios-evidence-paper",
          "locator": {
            "note": "Task definition, total count, creator protocol, and result row.",
            "type": "section",
            "value": "Table 1; Appendix Tables 2–6; task description for CloningScenarios"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/kind",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-cloning-scenarios-evidence-repository",
          "locator": {
            "note": "Public/private/total 33/8/41.",
            "type": "repository-path",
            "value": "CloningScenarios/cloningscenarios-v1-splits.json; CloningScenarios/cloningscenarios-v1-public.jsonl where present; task.py at 998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "source_id": "lab-bench-cloning-scenarios-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets"
          ]
        }
      ],
      "field_status": [],
      "id": "lab-bench-cloning-scenarios",
      "implementations": [
        {
          "commit": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838",
          "framework": "labbench Python evaluator",
          "notes": "Official task loader and exact-choice evaluator integration.",
          "status": "official",
          "url": "https://github.com/Future-House/LAB-Bench/blob/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/CloningScenarios/task.py"
        }
      ],
      "kind": "track",
      "latest_version": "repository-998a8e0",
      "modalities": [
        "text",
        "dna-rna-sequence"
      ],
      "name": "LAB-Bench CloningScenarios",
      "organizations": [
        "FutureHouse"
      ],
      "parent_id": "lab-bench",
      "release_date": "2024-07-14",
      "resources": [
        {
          "access_notes": "Commit-pinned official task files with prefix CloningScenarios/cloningscenarios-v1.",
          "id": "lab-bench-cloning-scenarios-repository-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-SA-4.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/CloningScenarios",
            "value": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "type": "repository",
          "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/CloningScenarios"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "repository-998a8e0",
        "entries": [
          {
            "confidence": "high",
            "count": 41,
            "count_basis": "questions across public and private splits",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-cloning-scenarios-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Molecular-cloning workflow planning.",
            "reporting_status": "reported",
            "task_type_id": "experiment-protocol-planning"
          }
        ],
        "notes": "Single-purpose formal LAB-Bench track.",
        "status": "complete"
      },
      "summary": "Human-hard, multi-step multiple-choice scenarios involving plasmids, DNA fragments, enzymes, and molecular-cloning workflows.",
      "task_counts": {
        "basis": "questions across public and private splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "questions",
            "count": 33,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-cloning-scenarios-public",
            "label": "Public split",
            "notes": "Released in the official repository and Hugging Face dataset.",
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 8,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-cloning-scenarios-private",
            "label": "Private contamination-monitoring split",
            "notes": "IDs are published; question contents are withheld.",
            "reporting_status": "reported"
          },
          {
            "basis": "modified questions sampled for the creator open-response study",
            "count": 10,
            "exclusive": false,
            "exhaustive": false,
            "id": "lab-bench-cloning-scenarios-creator-open-response",
            "label": "Creator open-response study",
            "notes": "A non-exhaustive derived subset; wording was modified to remove multiple-choice framing.",
            "reporting_status": "reported"
          }
        ],
        "total": 41
      },
      "task_formats": [
        "multiple choice"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task count and public/private membership independently recomputed from the pinned split manifest.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2024-07-17",
          "evidence_ids": [
            "lab-bench-cloning-scenarios-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "lab-bench-cloning-scenarios-paper-v3",
          "label": "paper-v3",
          "notes": "Creator-paper full task snapshot used for Tables 2–4.",
          "release_date": "2024-07-17",
          "status": "active",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 33,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-cloning-scenarios-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 8,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-cloning-scenarios-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              },
              {
                "basis": "modified questions sampled for the creator open-response study",
                "count": 10,
                "exclusive": false,
                "exhaustive": false,
                "id": "lab-bench-cloning-scenarios-creator-open-response",
                "label": "Creator open-response study",
                "notes": "A non-exhaustive derived subset; wording was modified to remove multiple-choice framing.",
                "reporting_status": "reported"
              }
            ],
            "total": 41
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "lab-bench-cloning-scenarios-evidence-repository"
          ],
          "formal_tracks": [],
          "id": "lab-bench-cloning-scenarios-repository-998a8e0",
          "label": "repository-998a8e0",
          "notes": "Current commit-pinned task snapshot.",
          "release_date": "2025-09-27",
          "status": "current",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 33,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-cloning-scenarios-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 8,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-cloning-scenarios-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              },
              {
                "basis": "modified questions sampled for the creator open-response study",
                "count": 10,
                "exclusive": false,
                "exhaustive": false,
                "id": "lab-bench-cloning-scenarios-creator-open-response",
                "label": "Creator open-response study",
                "notes": "A non-exhaustive derived subset; wording was modified to remove multiple-choice framing.",
                "reporting_status": "reported"
              }
            ],
            "total": 41
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "Public questions and all public/private split IDs are versioned in the official repository.",
        "biosafety_notes": "BioBench Atlas stores metadata and aggregate results only.",
        "grader": "The official harness scores exact single-choice outputs and reports accuracy, precision, coverage, and n.",
        "level": "partially-open",
        "license": "CC-BY-SA-4.0",
        "tasks": "520 questions are public and 130 question contents are withheld."
      },
      "aliases": [
        "DbQA"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": null,
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "retrieval",
        "knowledge",
        "prediction",
        "classification",
        "scientific-reasoning"
      ],
      "coverage_notes": [
        {
          "count": 200,
          "coverage": "explicitly-in-scope",
          "notes": "Two ClinVar protein-sequence tasks contain 100 questions each.",
          "reporting_status": "reported",
          "tag": "protein-sequence"
        },
        {
          "count": 50,
          "coverage": "explicitly-in-scope",
          "notes": "Viral PPI contains 50 database-retrieval questions.",
          "reporting_status": "reported",
          "tag": "protein-protein-binding"
        }
      ],
      "domains": [
        "bioinformatics",
        "genomics",
        "transcriptomics",
        "epigenomics",
        "protein-sequence",
        "protein-protein-binding",
        "clinical-translational",
        "virology"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-evidence-paper",
          "locator": {
            "note": "Category total, task definitions, domains, modalities, and creator evaluation.",
            "type": "table",
            "value": "Table 1 and Appendix Table 6; task-specific Appendix C sections"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/kind",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/license",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-evidence-repository",
          "locator": {
            "note": "Public/private/total 520/130/650; exact child files and evaluator.",
            "type": "repository-path",
            "value": "DbQA/*-splits.json and task.py at 998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "source_id": "lab-bench-dbqa-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        }
      ],
      "field_status": [],
      "id": "lab-bench-dbqa",
      "implementations": [
        {
          "commit": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838",
          "framework": "labbench Python evaluator",
          "notes": "Official category loader and evaluator.",
          "status": "official",
          "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
        }
      ],
      "kind": "track",
      "latest_version": "repository-998a8e0",
      "modalities": [
        "text",
        "protein-sequence",
        "database",
        "web"
      ],
      "name": "LAB-Bench DbQA",
      "organizations": [
        "FutureHouse"
      ],
      "parent_id": "lab-bench",
      "release_date": "2024-07-14",
      "resources": [
        {
          "access_notes": "Commit-pinned official category directory.",
          "id": "lab-bench-dbqa-repository-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-SA-4.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/DbQA",
            "value": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "type": "repository",
          "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/DbQA"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "repository-998a8e0",
        "entries": [
          {
            "confidence": "high",
            "count": 650,
            "count_basis": "questions across formal child tasks",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-dbqa-evidence-repository"
            ],
            "mapping_method": "official-track",
            "notes": "Complete formal DbQA category.",
            "reporting_status": "reported",
            "task_type_id": "scientific-database-retrieval"
          }
        ],
        "notes": "Single-purpose formal LAB-Bench track.",
        "status": "complete"
      },
      "summary": "Database-retrieval category spanning 10 genomics, clinical, protein, regulatory, vaccine-response, and viral-PPI tasks.",
      "task_counts": {
        "basis": "questions across formal child tasks",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "questions",
            "count": 50,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-dbqa-dga",
            "label": "Disease gene associations",
            "notes": null,
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 50,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-dbqa-gene-location",
            "label": "Gene location",
            "notes": null,
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 50,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-dbqa-mirna-targets",
            "label": "miRNA targets",
            "notes": null,
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 100,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-dbqa-mouse-tumor-gene-sets",
            "label": "Mouse tumor gene sets",
            "notes": null,
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 50,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-dbqa-oncogenic-signatures",
            "label": "Oncogenic signatures",
            "notes": null,
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 50,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-dbqa-tfbs-gtrd",
            "label": "GTRD transcription-factor binding sites",
            "notes": null,
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 100,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-dbqa-variant-from-sequence",
            "label": "Protein variant from sequence",
            "notes": null,
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 100,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-dbqa-variant-multi-sequence",
            "label": "Protein variant with multiple sequences",
            "notes": null,
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 50,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-dbqa-vax-response",
            "label": "Vaccine response gene sets",
            "notes": null,
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 50,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-dbqa-viral-ppi",
            "label": "Viral protein–protein interactions",
            "notes": null,
            "reporting_status": "reported"
          }
        ],
        "total": 650
      },
      "task_formats": [
        "multiple choice"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Counts and child-task relationships independently recomputed from versioned split manifests.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2024-07-17",
          "evidence_ids": [
            "lab-bench-dbqa-evidence-paper"
          ],
          "formal_tracks": [
            "lab-bench-dbqa-dga",
            "lab-bench-dbqa-gene-location",
            "lab-bench-dbqa-mirna-targets",
            "lab-bench-dbqa-mouse-tumor-gene-sets",
            "lab-bench-dbqa-oncogenic-signatures",
            "lab-bench-dbqa-tfbs-gtrd",
            "lab-bench-dbqa-variant-from-sequence",
            "lab-bench-dbqa-variant-multi-sequence",
            "lab-bench-dbqa-vax-response",
            "lab-bench-dbqa-viral-ppi"
          ],
          "id": "lab-bench-dbqa-paper-v3",
          "label": "paper-v3",
          "notes": "Creator-paper full category snapshot.",
          "release_date": "2024-07-17",
          "status": "active",
          "task_counts": {
            "basis": "questions across formal child tasks",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-dga",
                "label": "Disease gene associations",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-gene-location",
                "label": "Gene location",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-mirna-targets",
                "label": "miRNA targets",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 100,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-mouse-tumor-gene-sets",
                "label": "Mouse tumor gene sets",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-oncogenic-signatures",
                "label": "Oncogenic signatures",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-tfbs-gtrd",
                "label": "GTRD transcription-factor binding sites",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 100,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-variant-from-sequence",
                "label": "Protein variant from sequence",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 100,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-variant-multi-sequence",
                "label": "Protein variant with multiple sequences",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-vax-response",
                "label": "Vaccine response gene sets",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-viral-ppi",
                "label": "Viral protein–protein interactions",
                "notes": null,
                "reporting_status": "reported"
              }
            ],
            "total": 650
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "lab-bench-dbqa-evidence-repository"
          ],
          "formal_tracks": [
            "lab-bench-dbqa-dga",
            "lab-bench-dbqa-gene-location",
            "lab-bench-dbqa-mirna-targets",
            "lab-bench-dbqa-mouse-tumor-gene-sets",
            "lab-bench-dbqa-oncogenic-signatures",
            "lab-bench-dbqa-tfbs-gtrd",
            "lab-bench-dbqa-variant-from-sequence",
            "lab-bench-dbqa-variant-multi-sequence",
            "lab-bench-dbqa-vax-response",
            "lab-bench-dbqa-viral-ppi"
          ],
          "id": "lab-bench-dbqa-repository-998a8e0",
          "label": "repository-998a8e0",
          "notes": "Current commit-pinned category snapshot.",
          "release_date": "2025-09-27",
          "status": "current",
          "task_counts": {
            "basis": "questions across formal child tasks",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-dga",
                "label": "Disease gene associations",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-gene-location",
                "label": "Gene location",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-mirna-targets",
                "label": "miRNA targets",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 100,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-mouse-tumor-gene-sets",
                "label": "Mouse tumor gene sets",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-oncogenic-signatures",
                "label": "Oncogenic signatures",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-tfbs-gtrd",
                "label": "GTRD transcription-factor binding sites",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 100,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-variant-from-sequence",
                "label": "Protein variant from sequence",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 100,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-variant-multi-sequence",
                "label": "Protein variant with multiple sequences",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-vax-response",
                "label": "Vaccine response gene sets",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-viral-ppi",
                "label": "Viral protein–protein interactions",
                "notes": null,
                "reporting_status": "reported"
              }
            ],
            "total": 650
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official repository publishes the public questions and both public/private ID manifests.",
        "biosafety_notes": "BioBench Atlas stores metadata and aggregate results only and does not mirror question or sequence content.",
        "grader": "The official harness scores exact single-choice outputs and reports accuracy, precision, coverage, and n.",
        "level": "partially-open",
        "license": "CC-BY-SA-4.0",
        "tasks": "40 questions are public and 10 question contents are withheld."
      },
      "aliases": [
        "DbQA — Disease gene associations"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": null,
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "retrieval",
        "knowledge",
        "scientific-reasoning"
      ],
      "coverage_notes": [],
      "domains": [
        "genomics",
        "clinical-translational",
        "bioinformatics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-dbqa-dga-creator-mcq"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-dga-evidence-paper",
          "locator": {
            "note": "Task definition, total count, creator protocol, and result row.",
            "type": "section",
            "value": "Table 1; Appendix Tables 2–6; task description for DbQA_dga_task-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/kind",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-dga-evidence-repository",
          "locator": {
            "note": "Public/private/total 40/10/50.",
            "type": "repository-path",
            "value": "DbQA/dga_task-v1-splits.json; DbQA/dga_task-v1-public.jsonl where present; task.py at 998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "source_id": "lab-bench-dbqa-dga-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets"
          ]
        }
      ],
      "field_status": [],
      "id": "lab-bench-dbqa-dga",
      "implementations": [
        {
          "commit": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838",
          "framework": "labbench Python evaluator",
          "notes": "Official task loader and exact-choice evaluator integration.",
          "status": "official",
          "url": "https://github.com/Future-House/LAB-Bench/blob/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/DbQA/task.py"
        }
      ],
      "kind": "track",
      "latest_version": "repository-998a8e0",
      "modalities": [
        "text",
        "database",
        "web"
      ],
      "name": "LAB-Bench DbQA — Disease gene associations",
      "organizations": [
        "FutureHouse"
      ],
      "parent_id": "lab-bench-dbqa",
      "release_date": "2024-07-14",
      "resources": [
        {
          "access_notes": "Commit-pinned official task files with prefix DbQA/dga_task-v1.",
          "id": "lab-bench-dbqa-dga-repository-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-SA-4.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/DbQA",
            "value": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "type": "repository",
          "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/DbQA"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "repository-998a8e0",
        "entries": [
          {
            "confidence": "high",
            "count": 50,
            "count_basis": "questions across public and private splits",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-dbqa-dga-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "DisGeNET and OMIM retrieval.",
            "reporting_status": "reported",
            "task_type_id": "scientific-database-retrieval"
          }
        ],
        "notes": "Single-purpose formal LAB-Bench track.",
        "status": "complete"
      },
      "summary": "Identifies genes associated with a phenotype in DisGeNET but not OMIM.",
      "task_counts": {
        "basis": "questions across public and private splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "questions",
            "count": 40,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-dbqa-dga-public",
            "label": "Public split",
            "notes": "Released in the official repository and Hugging Face dataset.",
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 10,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-dbqa-dga-private",
            "label": "Private contamination-monitoring split",
            "notes": "IDs are published; question contents are withheld.",
            "reporting_status": "reported"
          }
        ],
        "total": 50
      },
      "task_formats": [
        "multiple choice"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task count and public/private membership independently recomputed from the pinned split manifest.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2024-07-17",
          "evidence_ids": [
            "lab-bench-dbqa-dga-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "lab-bench-dbqa-dga-paper-v3",
          "label": "paper-v3",
          "notes": "Creator-paper full task snapshot used for Tables 2–4.",
          "release_date": "2024-07-17",
          "status": "active",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-dga-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-dga-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "lab-bench-dbqa-dga-evidence-repository"
          ],
          "formal_tracks": [],
          "id": "lab-bench-dbqa-dga-repository-998a8e0",
          "label": "repository-998a8e0",
          "notes": "Current commit-pinned task snapshot.",
          "release_date": "2025-09-27",
          "status": "current",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-dga-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-dga-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official repository publishes the public questions and both public/private ID manifests.",
        "biosafety_notes": "BioBench Atlas stores metadata and aggregate results only and does not mirror question or sequence content.",
        "grader": "The official harness scores exact single-choice outputs and reports accuracy, precision, coverage, and n.",
        "level": "partially-open",
        "license": "CC-BY-SA-4.0",
        "tasks": "40 questions are public and 10 question contents are withheld."
      },
      "aliases": [
        "DbQA — Gene location"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": null,
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "retrieval",
        "knowledge"
      ],
      "coverage_notes": [],
      "domains": [
        "genomics",
        "bioinformatics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-dbqa-gene-location-creator-mcq"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-gene-location-evidence-paper",
          "locator": {
            "note": "Task definition, total count, creator protocol, and result row.",
            "type": "section",
            "value": "Table 1; Appendix Tables 2–6; task description for DbQA_gene_location_task-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/kind",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-gene-location-evidence-repository",
          "locator": {
            "note": "Public/private/total 40/10/50.",
            "type": "repository-path",
            "value": "DbQA/gene_location_task-v1-splits.json; DbQA/gene_location_task-v1-public.jsonl where present; task.py at 998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "source_id": "lab-bench-dbqa-gene-location-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets"
          ]
        }
      ],
      "field_status": [],
      "id": "lab-bench-dbqa-gene-location",
      "implementations": [
        {
          "commit": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838",
          "framework": "labbench Python evaluator",
          "notes": "Official task loader and exact-choice evaluator integration.",
          "status": "official",
          "url": "https://github.com/Future-House/LAB-Bench/blob/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/DbQA/task.py"
        }
      ],
      "kind": "track",
      "latest_version": "repository-998a8e0",
      "modalities": [
        "text",
        "database",
        "web"
      ],
      "name": "LAB-Bench DbQA — Gene location",
      "organizations": [
        "FutureHouse"
      ],
      "parent_id": "lab-bench-dbqa",
      "release_date": "2024-07-14",
      "resources": [
        {
          "access_notes": "Commit-pinned official task files with prefix DbQA/gene_location_task-v1.",
          "id": "lab-bench-dbqa-gene-location-repository-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-SA-4.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/DbQA",
            "value": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "type": "repository",
          "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/DbQA"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "repository-998a8e0",
        "entries": [
          {
            "confidence": "high",
            "count": 50,
            "count_basis": "questions across public and private splits",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-dbqa-gene-location-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Ensembl gene-location retrieval.",
            "reporting_status": "reported",
            "task_type_id": "scientific-database-retrieval"
          }
        ],
        "notes": "Single-purpose formal LAB-Bench track.",
        "status": "complete"
      },
      "summary": "Retrieves human-gene cytogenetic locations from the stated Ensembl release.",
      "task_counts": {
        "basis": "questions across public and private splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "questions",
            "count": 40,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-dbqa-gene-location-public",
            "label": "Public split",
            "notes": "Released in the official repository and Hugging Face dataset.",
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 10,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-dbqa-gene-location-private",
            "label": "Private contamination-monitoring split",
            "notes": "IDs are published; question contents are withheld.",
            "reporting_status": "reported"
          }
        ],
        "total": 50
      },
      "task_formats": [
        "multiple choice"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task count and public/private membership independently recomputed from the pinned split manifest.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2024-07-17",
          "evidence_ids": [
            "lab-bench-dbqa-gene-location-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "lab-bench-dbqa-gene-location-paper-v3",
          "label": "paper-v3",
          "notes": "Creator-paper full task snapshot used for Tables 2–4.",
          "release_date": "2024-07-17",
          "status": "active",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-gene-location-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-gene-location-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "lab-bench-dbqa-gene-location-evidence-repository"
          ],
          "formal_tracks": [],
          "id": "lab-bench-dbqa-gene-location-repository-998a8e0",
          "label": "repository-998a8e0",
          "notes": "Current commit-pinned task snapshot.",
          "release_date": "2025-09-27",
          "status": "current",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-gene-location-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-gene-location-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official repository publishes the public questions and both public/private ID manifests.",
        "biosafety_notes": "BioBench Atlas stores metadata and aggregate results only and does not mirror question or sequence content.",
        "grader": "The official harness scores exact single-choice outputs and reports accuracy, precision, coverage, and n.",
        "level": "partially-open",
        "license": "CC-BY-SA-4.0",
        "tasks": "40 questions are public and 10 question contents are withheld."
      },
      "aliases": [
        "DbQA — miRNA targets"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": null,
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "retrieval",
        "prediction",
        "knowledge"
      ],
      "coverage_notes": [],
      "domains": [
        "transcriptomics",
        "genomics",
        "bioinformatics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-dbqa-mirna-targets-creator-mcq"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-mirna-targets-evidence-paper",
          "locator": {
            "note": "Task definition, total count, creator protocol, and result row.",
            "type": "section",
            "value": "Table 1; Appendix Tables 2–6; task description for DbQA_mirna_targets_task-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/kind",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-mirna-targets-evidence-repository",
          "locator": {
            "note": "Public/private/total 40/10/50.",
            "type": "repository-path",
            "value": "DbQA/mirna_targets_task-v1-splits.json; DbQA/mirna_targets_task-v1-public.jsonl where present; task.py at 998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "source_id": "lab-bench-dbqa-mirna-targets-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets"
          ]
        }
      ],
      "field_status": [],
      "id": "lab-bench-dbqa-mirna-targets",
      "implementations": [
        {
          "commit": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838",
          "framework": "labbench Python evaluator",
          "notes": "Official task loader and exact-choice evaluator integration.",
          "status": "official",
          "url": "https://github.com/Future-House/LAB-Bench/blob/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/DbQA/task.py"
        }
      ],
      "kind": "track",
      "latest_version": "repository-998a8e0",
      "modalities": [
        "text",
        "database",
        "web"
      ],
      "name": "LAB-Bench DbQA — miRNA targets",
      "organizations": [
        "FutureHouse"
      ],
      "parent_id": "lab-bench-dbqa",
      "release_date": "2024-07-14",
      "resources": [
        {
          "access_notes": "Commit-pinned official task files with prefix DbQA/mirna_targets_task-v1.",
          "id": "lab-bench-dbqa-mirna-targets-repository-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-SA-4.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/DbQA",
            "value": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "type": "repository",
          "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/DbQA"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "repository-998a8e0",
        "entries": [
          {
            "confidence": "high",
            "count": 50,
            "count_basis": "questions across public and private splits",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-dbqa-mirna-targets-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "miRDB target retrieval.",
            "reporting_status": "reported",
            "task_type_id": "scientific-database-retrieval"
          }
        ],
        "notes": "Single-purpose formal LAB-Bench track.",
        "status": "complete"
      },
      "summary": "Retrieves computationally predicted human miRNA targets from miRDB.",
      "task_counts": {
        "basis": "questions across public and private splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "questions",
            "count": 40,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-dbqa-mirna-targets-public",
            "label": "Public split",
            "notes": "Released in the official repository and Hugging Face dataset.",
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 10,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-dbqa-mirna-targets-private",
            "label": "Private contamination-monitoring split",
            "notes": "IDs are published; question contents are withheld.",
            "reporting_status": "reported"
          }
        ],
        "total": 50
      },
      "task_formats": [
        "multiple choice"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task count and public/private membership independently recomputed from the pinned split manifest.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2024-07-17",
          "evidence_ids": [
            "lab-bench-dbqa-mirna-targets-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "lab-bench-dbqa-mirna-targets-paper-v3",
          "label": "paper-v3",
          "notes": "Creator-paper full task snapshot used for Tables 2–4.",
          "release_date": "2024-07-17",
          "status": "active",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-mirna-targets-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-mirna-targets-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "lab-bench-dbqa-mirna-targets-evidence-repository"
          ],
          "formal_tracks": [],
          "id": "lab-bench-dbqa-mirna-targets-repository-998a8e0",
          "label": "repository-998a8e0",
          "notes": "Current commit-pinned task snapshot.",
          "release_date": "2025-09-27",
          "status": "current",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-mirna-targets-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-mirna-targets-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official repository publishes the public questions and both public/private ID manifests.",
        "biosafety_notes": "BioBench Atlas stores metadata and aggregate results only and does not mirror question or sequence content.",
        "grader": "The official harness scores exact single-choice outputs and reports accuracy, precision, coverage, and n.",
        "level": "partially-open",
        "license": "CC-BY-SA-4.0",
        "tasks": "80 questions are public and 20 question contents are withheld."
      },
      "aliases": [
        "DbQA — Mouse tumor gene sets"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": null,
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "retrieval",
        "knowledge"
      ],
      "coverage_notes": [],
      "domains": [
        "genomics",
        "clinical-translational",
        "bioinformatics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-dbqa-mouse-tumor-gene-sets-creator-mcq"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-mouse-tumor-gene-sets-evidence-paper",
          "locator": {
            "note": "Task definition, total count, creator protocol, and result row.",
            "type": "section",
            "value": "Table 1; Appendix Tables 2–6; task description for DbQA_mouse_tumor_gene_sets-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/kind",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-mouse-tumor-gene-sets-evidence-repository",
          "locator": {
            "note": "Public/private/total 80/20/100.",
            "type": "repository-path",
            "value": "DbQA/mouse_tumor_gene_sets-v1-splits.json; DbQA/mouse_tumor_gene_sets-v1-public.jsonl where present; task.py at 998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "source_id": "lab-bench-dbqa-mouse-tumor-gene-sets-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets"
          ]
        }
      ],
      "field_status": [],
      "id": "lab-bench-dbqa-mouse-tumor-gene-sets",
      "implementations": [
        {
          "commit": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838",
          "framework": "labbench Python evaluator",
          "notes": "Official task loader and exact-choice evaluator integration.",
          "status": "official",
          "url": "https://github.com/Future-House/LAB-Bench/blob/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/DbQA/task.py"
        }
      ],
      "kind": "track",
      "latest_version": "repository-998a8e0",
      "modalities": [
        "text",
        "database",
        "web"
      ],
      "name": "LAB-Bench DbQA — Mouse tumor gene sets",
      "organizations": [
        "FutureHouse"
      ],
      "parent_id": "lab-bench-dbqa",
      "release_date": "2024-07-14",
      "resources": [
        {
          "access_notes": "Commit-pinned official task files with prefix DbQA/mouse_tumor_gene_sets-v1.",
          "id": "lab-bench-dbqa-mouse-tumor-gene-sets-repository-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-SA-4.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/DbQA",
            "value": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "type": "repository",
          "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/DbQA"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "repository-998a8e0",
        "entries": [
          {
            "confidence": "high",
            "count": 100,
            "count_basis": "questions across public and private splits",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-dbqa-mouse-tumor-gene-sets-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Gene-set membership retrieval.",
            "reporting_status": "reported",
            "task_type_id": "scientific-database-retrieval"
          }
        ],
        "notes": "Single-purpose formal LAB-Bench track.",
        "status": "complete"
      },
      "summary": "Retrieves genes in Mammalian Phenotype Tumor Ontology gene sets.",
      "task_counts": {
        "basis": "questions across public and private splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "questions",
            "count": 80,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-dbqa-mouse-tumor-gene-sets-public",
            "label": "Public split",
            "notes": "Released in the official repository and Hugging Face dataset.",
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 20,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-dbqa-mouse-tumor-gene-sets-private",
            "label": "Private contamination-monitoring split",
            "notes": "IDs are published; question contents are withheld.",
            "reporting_status": "reported"
          }
        ],
        "total": 100
      },
      "task_formats": [
        "multiple choice"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task count and public/private membership independently recomputed from the pinned split manifest.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2024-07-17",
          "evidence_ids": [
            "lab-bench-dbqa-mouse-tumor-gene-sets-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "lab-bench-dbqa-mouse-tumor-gene-sets-paper-v3",
          "label": "paper-v3",
          "notes": "Creator-paper full task snapshot used for Tables 2–4.",
          "release_date": "2024-07-17",
          "status": "active",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 80,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-mouse-tumor-gene-sets-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 20,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-mouse-tumor-gene-sets-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 100
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "lab-bench-dbqa-mouse-tumor-gene-sets-evidence-repository"
          ],
          "formal_tracks": [],
          "id": "lab-bench-dbqa-mouse-tumor-gene-sets-repository-998a8e0",
          "label": "repository-998a8e0",
          "notes": "Current commit-pinned task snapshot.",
          "release_date": "2025-09-27",
          "status": "current",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 80,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-mouse-tumor-gene-sets-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 20,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-mouse-tumor-gene-sets-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 100
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official repository publishes the public questions and both public/private ID manifests.",
        "biosafety_notes": "BioBench Atlas stores metadata and aggregate results only and does not mirror question or sequence content.",
        "grader": "The official harness scores exact single-choice outputs and reports accuracy, precision, coverage, and n.",
        "level": "partially-open",
        "license": "CC-BY-SA-4.0",
        "tasks": "40 questions are public and 10 question contents are withheld."
      },
      "aliases": [
        "DbQA — Oncogenic signatures"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": null,
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "retrieval",
        "knowledge"
      ],
      "coverage_notes": [],
      "domains": [
        "genomics",
        "clinical-translational",
        "bioinformatics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-dbqa-oncogenic-signatures-creator-mcq"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-oncogenic-signatures-evidence-paper",
          "locator": {
            "note": "Task definition, total count, creator protocol, and result row.",
            "type": "section",
            "value": "Table 1; Appendix Tables 2–6; task description for DbQA_oncogenic_signatures_task-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/kind",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-oncogenic-signatures-evidence-repository",
          "locator": {
            "note": "Public/private/total 40/10/50.",
            "type": "repository-path",
            "value": "DbQA/oncogenic_signatures_task-v1-splits.json; DbQA/oncogenic_signatures_task-v1-public.jsonl where present; task.py at 998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "source_id": "lab-bench-dbqa-oncogenic-signatures-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets"
          ]
        }
      ],
      "field_status": [],
      "id": "lab-bench-dbqa-oncogenic-signatures",
      "implementations": [
        {
          "commit": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838",
          "framework": "labbench Python evaluator",
          "notes": "Official task loader and exact-choice evaluator integration.",
          "status": "official",
          "url": "https://github.com/Future-House/LAB-Bench/blob/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/DbQA/task.py"
        }
      ],
      "kind": "track",
      "latest_version": "repository-998a8e0",
      "modalities": [
        "text",
        "database",
        "web"
      ],
      "name": "LAB-Bench DbQA — Oncogenic signatures",
      "organizations": [
        "FutureHouse"
      ],
      "parent_id": "lab-bench-dbqa",
      "release_date": "2024-07-14",
      "resources": [
        {
          "access_notes": "Commit-pinned official task files with prefix DbQA/oncogenic_signatures_task-v1.",
          "id": "lab-bench-dbqa-oncogenic-signatures-repository-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-SA-4.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/DbQA",
            "value": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "type": "repository",
          "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/DbQA"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "repository-998a8e0",
        "entries": [
          {
            "confidence": "high",
            "count": 50,
            "count_basis": "questions across public and private splits",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-dbqa-oncogenic-signatures-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "MSigDB signature retrieval.",
            "reporting_status": "reported",
            "task_type_id": "scientific-database-retrieval"
          }
        ],
        "notes": "Single-purpose formal LAB-Bench track.",
        "status": "complete"
      },
      "summary": "Retrieves membership in MSigDB C6 oncogenic-signature gene sets.",
      "task_counts": {
        "basis": "questions across public and private splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "questions",
            "count": 40,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-dbqa-oncogenic-signatures-public",
            "label": "Public split",
            "notes": "Released in the official repository and Hugging Face dataset.",
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 10,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-dbqa-oncogenic-signatures-private",
            "label": "Private contamination-monitoring split",
            "notes": "IDs are published; question contents are withheld.",
            "reporting_status": "reported"
          }
        ],
        "total": 50
      },
      "task_formats": [
        "multiple choice"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task count and public/private membership independently recomputed from the pinned split manifest.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2024-07-17",
          "evidence_ids": [
            "lab-bench-dbqa-oncogenic-signatures-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "lab-bench-dbqa-oncogenic-signatures-paper-v3",
          "label": "paper-v3",
          "notes": "Creator-paper full task snapshot used for Tables 2–4.",
          "release_date": "2024-07-17",
          "status": "active",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-oncogenic-signatures-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-oncogenic-signatures-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "lab-bench-dbqa-oncogenic-signatures-evidence-repository"
          ],
          "formal_tracks": [],
          "id": "lab-bench-dbqa-oncogenic-signatures-repository-998a8e0",
          "label": "repository-998a8e0",
          "notes": "Current commit-pinned task snapshot.",
          "release_date": "2025-09-27",
          "status": "current",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-oncogenic-signatures-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-oncogenic-signatures-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official repository publishes the public questions and both public/private ID manifests.",
        "biosafety_notes": "BioBench Atlas stores metadata and aggregate results only and does not mirror question or sequence content.",
        "grader": "The official harness scores exact single-choice outputs and reports accuracy, precision, coverage, and n.",
        "level": "partially-open",
        "license": "CC-BY-SA-4.0",
        "tasks": "40 questions are public and 10 question contents are withheld."
      },
      "aliases": [
        "DbQA — GTRD transcription-factor binding sites"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": null,
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "retrieval",
        "knowledge"
      ],
      "coverage_notes": [],
      "domains": [
        "genomics",
        "epigenomics",
        "bioinformatics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-dbqa-tfbs-gtrd-creator-mcq"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-tfbs-gtrd-evidence-paper",
          "locator": {
            "note": "Task definition, total count, creator protocol, and result row.",
            "type": "section",
            "value": "Table 1; Appendix Tables 2–6; task description for DbQA_tfbs_GTRD_task-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/kind",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0",
            "/scientific_task_classification/entries/1"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-tfbs-gtrd-evidence-repository",
          "locator": {
            "note": "Public/private/total 40/10/50.",
            "type": "repository-path",
            "value": "DbQA/tfbs_GTRD_task-v1-splits.json; DbQA/tfbs_GTRD_task-v1-public.jsonl where present; task.py at 998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "source_id": "lab-bench-dbqa-tfbs-gtrd-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets"
          ]
        }
      ],
      "field_status": [],
      "id": "lab-bench-dbqa-tfbs-gtrd",
      "implementations": [
        {
          "commit": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838",
          "framework": "labbench Python evaluator",
          "notes": "Official task loader and exact-choice evaluator integration.",
          "status": "official",
          "url": "https://github.com/Future-House/LAB-Bench/blob/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/DbQA/task.py"
        }
      ],
      "kind": "track",
      "latest_version": "repository-998a8e0",
      "modalities": [
        "text",
        "database",
        "web"
      ],
      "name": "LAB-Bench DbQA — GTRD transcription-factor binding sites",
      "organizations": [
        "FutureHouse"
      ],
      "parent_id": "lab-bench-dbqa",
      "release_date": "2024-07-14",
      "resources": [
        {
          "access_notes": "Commit-pinned official task files with prefix DbQA/tfbs_GTRD_task-v1.",
          "id": "lab-bench-dbqa-tfbs-gtrd-repository-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-SA-4.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/DbQA",
            "value": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "type": "repository",
          "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/DbQA"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "repository-998a8e0",
        "entries": [
          {
            "confidence": "high",
            "count": 50,
            "count_basis": "questions across public and private splits",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-dbqa-tfbs-gtrd-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Database retrieval is the evaluated operation.",
            "reporting_status": "reported",
            "task_type_id": "scientific-database-retrieval"
          },
          {
            "confidence": "high",
            "count": 50,
            "count_basis": "questions across public and private splits",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "observed",
            "evidence_ids": [
              "lab-bench-dbqa-tfbs-gtrd-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "The artifact concerns TFBS annotations; it does not evaluate a de novo predictor.",
            "reporting_status": "reported",
            "task_type_id": "transcription-factor-binding-site-prediction"
          }
        ],
        "notes": "Official GTRD retrieval track with a regulatory scientific object.",
        "status": "complete"
      },
      "summary": "Retrieves promoter-region transcription-factor binding-site annotations from GTRD.",
      "task_counts": {
        "basis": "questions across public and private splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "questions",
            "count": 40,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-dbqa-tfbs-gtrd-public",
            "label": "Public split",
            "notes": "Released in the official repository and Hugging Face dataset.",
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 10,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-dbqa-tfbs-gtrd-private",
            "label": "Private contamination-monitoring split",
            "notes": "IDs are published; question contents are withheld.",
            "reporting_status": "reported"
          }
        ],
        "total": 50
      },
      "task_formats": [
        "multiple choice"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task count and public/private membership independently recomputed from the pinned split manifest.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2024-07-17",
          "evidence_ids": [
            "lab-bench-dbqa-tfbs-gtrd-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "lab-bench-dbqa-tfbs-gtrd-paper-v3",
          "label": "paper-v3",
          "notes": "Creator-paper full task snapshot used for Tables 2–4.",
          "release_date": "2024-07-17",
          "status": "active",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-tfbs-gtrd-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-tfbs-gtrd-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "lab-bench-dbqa-tfbs-gtrd-evidence-repository"
          ],
          "formal_tracks": [],
          "id": "lab-bench-dbqa-tfbs-gtrd-repository-998a8e0",
          "label": "repository-998a8e0",
          "notes": "Current commit-pinned task snapshot.",
          "release_date": "2025-09-27",
          "status": "current",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-tfbs-gtrd-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-tfbs-gtrd-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official repository publishes the public questions and both public/private ID manifests.",
        "biosafety_notes": "BioBench Atlas stores metadata and aggregate results only and does not mirror question or sequence content.",
        "grader": "The official harness scores exact single-choice outputs and reports accuracy, precision, coverage, and n.",
        "level": "partially-open",
        "license": "CC-BY-SA-4.0",
        "tasks": "80 questions are public and 20 question contents are withheld."
      },
      "aliases": [
        "DbQA — Protein variant from sequence"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": null,
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "retrieval",
        "classification",
        "scientific-reasoning"
      ],
      "coverage_notes": [
        {
          "count": 100,
          "coverage": "explicitly-in-scope",
          "notes": "All questions contain or compare protein sequences.",
          "reporting_status": "reported",
          "tag": "protein-sequence"
        }
      ],
      "domains": [
        "protein-sequence",
        "clinical-translational",
        "bioinformatics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-dbqa-variant-from-sequence-creator-mcq"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-variant-from-sequence-evidence-paper",
          "locator": {
            "note": "Task definition, total count, creator protocol, and result row.",
            "type": "section",
            "value": "Table 1; Appendix Tables 2–6; task description for DbQA_variant_from_sequence_task-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/kind",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0",
            "/scientific_task_classification/entries/1"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-variant-from-sequence-evidence-repository",
          "locator": {
            "note": "Public/private/total 80/20/100.",
            "type": "repository-path",
            "value": "DbQA/variant_from_sequence_task-v1-splits.json; DbQA/variant_from_sequence_task-v1-public.jsonl where present; task.py at 998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "source_id": "lab-bench-dbqa-variant-from-sequence-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets"
          ]
        }
      ],
      "field_status": [],
      "id": "lab-bench-dbqa-variant-from-sequence",
      "implementations": [
        {
          "commit": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838",
          "framework": "labbench Python evaluator",
          "notes": "Official task loader and exact-choice evaluator integration.",
          "status": "official",
          "url": "https://github.com/Future-House/LAB-Bench/blob/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/DbQA/task.py"
        }
      ],
      "kind": "track",
      "latest_version": "repository-998a8e0",
      "modalities": [
        "text",
        "protein-sequence",
        "database",
        "web"
      ],
      "name": "LAB-Bench DbQA — Protein variant from sequence",
      "organizations": [
        "FutureHouse"
      ],
      "parent_id": "lab-bench-dbqa",
      "release_date": "2024-07-14",
      "resources": [
        {
          "access_notes": "Commit-pinned official task files with prefix DbQA/variant_from_sequence_task-v1.",
          "id": "lab-bench-dbqa-variant-from-sequence-repository-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-SA-4.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/DbQA",
            "value": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "type": "repository",
          "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/DbQA"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "repository-998a8e0",
        "entries": [
          {
            "confidence": "high",
            "count": 100,
            "count_basis": "questions across public and private splits",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-dbqa-variant-from-sequence-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "ClinVar retrieval.",
            "reporting_status": "reported",
            "task_type_id": "scientific-database-retrieval"
          },
          {
            "confidence": "high",
            "count": 100,
            "count_basis": "questions across public and private splits",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-dbqa-variant-from-sequence-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Variant pathogenicity is explicitly evaluated.",
            "reporting_status": "reported",
            "task_type_id": "protein-clinical-variant-interpretation"
          }
        ],
        "notes": "ClinVar lookup and pathogenicity interpretation from a protein sequence.",
        "status": "complete"
      },
      "summary": "Uses a protein sequence and ClinVar lookup to identify benign or pathogenic variants.",
      "task_counts": {
        "basis": "questions across public and private splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "questions",
            "count": 80,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-dbqa-variant-from-sequence-public",
            "label": "Public split",
            "notes": "Released in the official repository and Hugging Face dataset.",
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 20,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-dbqa-variant-from-sequence-private",
            "label": "Private contamination-monitoring split",
            "notes": "IDs are published; question contents are withheld.",
            "reporting_status": "reported"
          }
        ],
        "total": 100
      },
      "task_formats": [
        "multiple choice"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task count and public/private membership independently recomputed from the pinned split manifest.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2024-07-17",
          "evidence_ids": [
            "lab-bench-dbqa-variant-from-sequence-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "lab-bench-dbqa-variant-from-sequence-paper-v3",
          "label": "paper-v3",
          "notes": "Creator-paper full task snapshot used for Tables 2–4.",
          "release_date": "2024-07-17",
          "status": "active",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 80,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-variant-from-sequence-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 20,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-variant-from-sequence-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 100
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "lab-bench-dbqa-variant-from-sequence-evidence-repository"
          ],
          "formal_tracks": [],
          "id": "lab-bench-dbqa-variant-from-sequence-repository-998a8e0",
          "label": "repository-998a8e0",
          "notes": "Current commit-pinned task snapshot.",
          "release_date": "2025-09-27",
          "status": "current",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 80,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-variant-from-sequence-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 20,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-variant-from-sequence-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 100
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official repository publishes the public questions and both public/private ID manifests.",
        "biosafety_notes": "BioBench Atlas stores metadata and aggregate results only and does not mirror question or sequence content.",
        "grader": "The official harness scores exact single-choice outputs and reports accuracy, precision, coverage, and n.",
        "level": "partially-open",
        "license": "CC-BY-SA-4.0",
        "tasks": "80 questions are public and 20 question contents are withheld."
      },
      "aliases": [
        "DbQA — Protein variant with multiple sequences"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": null,
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "retrieval",
        "classification",
        "scientific-reasoning"
      ],
      "coverage_notes": [
        {
          "count": 100,
          "coverage": "explicitly-in-scope",
          "notes": "All questions contain or compare protein sequences.",
          "reporting_status": "reported",
          "tag": "protein-sequence"
        }
      ],
      "domains": [
        "protein-sequence",
        "clinical-translational",
        "bioinformatics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-dbqa-variant-multi-sequence-creator-mcq"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-variant-multi-sequence-evidence-paper",
          "locator": {
            "note": "Task definition, total count, creator protocol, and result row.",
            "type": "section",
            "value": "Table 1; Appendix Tables 2–6; task description for DbQA_variant_multi_sequence_task-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/kind",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0",
            "/scientific_task_classification/entries/1"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-variant-multi-sequence-evidence-repository",
          "locator": {
            "note": "Public/private/total 80/20/100.",
            "type": "repository-path",
            "value": "DbQA/variant_multi_sequence_task-v1-splits.json; DbQA/variant_multi_sequence_task-v1-public.jsonl where present; task.py at 998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "source_id": "lab-bench-dbqa-variant-multi-sequence-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets"
          ]
        }
      ],
      "field_status": [],
      "id": "lab-bench-dbqa-variant-multi-sequence",
      "implementations": [
        {
          "commit": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838",
          "framework": "labbench Python evaluator",
          "notes": "Official task loader and exact-choice evaluator integration.",
          "status": "official",
          "url": "https://github.com/Future-House/LAB-Bench/blob/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/DbQA/task.py"
        }
      ],
      "kind": "track",
      "latest_version": "repository-998a8e0",
      "modalities": [
        "text",
        "protein-sequence",
        "database",
        "web"
      ],
      "name": "LAB-Bench DbQA — Protein variant with multiple sequences",
      "organizations": [
        "FutureHouse"
      ],
      "parent_id": "lab-bench-dbqa",
      "release_date": "2024-07-14",
      "resources": [
        {
          "access_notes": "Commit-pinned official task files with prefix DbQA/variant_multi_sequence_task-v1.",
          "id": "lab-bench-dbqa-variant-multi-sequence-repository-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-SA-4.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/DbQA",
            "value": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "type": "repository",
          "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/DbQA"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "repository-998a8e0",
        "entries": [
          {
            "confidence": "high",
            "count": 100,
            "count_basis": "questions across public and private splits",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-dbqa-variant-multi-sequence-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "ClinVar retrieval.",
            "reporting_status": "reported",
            "task_type_id": "scientific-database-retrieval"
          },
          {
            "confidence": "high",
            "count": 100,
            "count_basis": "questions across public and private splits",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-dbqa-variant-multi-sequence-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Variant pathogenicity is explicitly evaluated.",
            "reporting_status": "reported",
            "task_type_id": "protein-clinical-variant-interpretation"
          }
        ],
        "notes": "ClinVar lookup and pathogenicity interpretation across multiple protein sequences.",
        "status": "complete"
      },
      "summary": "Identifies ClinVar variant pathogenicity while reasoning across multiple protein sequences.",
      "task_counts": {
        "basis": "questions across public and private splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "questions",
            "count": 80,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-dbqa-variant-multi-sequence-public",
            "label": "Public split",
            "notes": "Released in the official repository and Hugging Face dataset.",
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 20,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-dbqa-variant-multi-sequence-private",
            "label": "Private contamination-monitoring split",
            "notes": "IDs are published; question contents are withheld.",
            "reporting_status": "reported"
          }
        ],
        "total": 100
      },
      "task_formats": [
        "multiple choice"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task count and public/private membership independently recomputed from the pinned split manifest.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2024-07-17",
          "evidence_ids": [
            "lab-bench-dbqa-variant-multi-sequence-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "lab-bench-dbqa-variant-multi-sequence-paper-v3",
          "label": "paper-v3",
          "notes": "Creator-paper full task snapshot used for Tables 2–4.",
          "release_date": "2024-07-17",
          "status": "active",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 80,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-variant-multi-sequence-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 20,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-variant-multi-sequence-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 100
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "lab-bench-dbqa-variant-multi-sequence-evidence-repository"
          ],
          "formal_tracks": [],
          "id": "lab-bench-dbqa-variant-multi-sequence-repository-998a8e0",
          "label": "repository-998a8e0",
          "notes": "Current commit-pinned task snapshot.",
          "release_date": "2025-09-27",
          "status": "current",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 80,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-variant-multi-sequence-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 20,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-variant-multi-sequence-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 100
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official repository publishes the public questions and both public/private ID manifests.",
        "biosafety_notes": "BioBench Atlas stores metadata and aggregate results only and does not mirror question or sequence content.",
        "grader": "The official harness scores exact single-choice outputs and reports accuracy, precision, coverage, and n.",
        "level": "partially-open",
        "license": "CC-BY-SA-4.0",
        "tasks": "40 questions are public and 10 question contents are withheld."
      },
      "aliases": [
        "DbQA — Vaccine response gene sets"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": null,
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "retrieval",
        "knowledge"
      ],
      "coverage_notes": [],
      "domains": [
        "transcriptomics",
        "clinical-translational",
        "bioinformatics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-dbqa-vax-response-creator-mcq"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-vax-response-evidence-paper",
          "locator": {
            "note": "Task definition, total count, creator protocol, and result row.",
            "type": "section",
            "value": "Table 1; Appendix Tables 2–6; task description for DbQA_vax_response_task-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/kind",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-vax-response-evidence-repository",
          "locator": {
            "note": "Public/private/total 40/10/50.",
            "type": "repository-path",
            "value": "DbQA/vax_response_task-v1-splits.json; DbQA/vax_response_task-v1-public.jsonl where present; task.py at 998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "source_id": "lab-bench-dbqa-vax-response-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets"
          ]
        }
      ],
      "field_status": [],
      "id": "lab-bench-dbqa-vax-response",
      "implementations": [
        {
          "commit": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838",
          "framework": "labbench Python evaluator",
          "notes": "Official task loader and exact-choice evaluator integration.",
          "status": "official",
          "url": "https://github.com/Future-House/LAB-Bench/blob/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/DbQA/task.py"
        }
      ],
      "kind": "track",
      "latest_version": "repository-998a8e0",
      "modalities": [
        "text",
        "database",
        "web"
      ],
      "name": "LAB-Bench DbQA — Vaccine response gene sets",
      "organizations": [
        "FutureHouse"
      ],
      "parent_id": "lab-bench-dbqa",
      "release_date": "2024-07-14",
      "resources": [
        {
          "access_notes": "Commit-pinned official task files with prefix DbQA/vax_response_task-v1.",
          "id": "lab-bench-dbqa-vax-response-repository-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-SA-4.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/DbQA",
            "value": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "type": "repository",
          "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/DbQA"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "repository-998a8e0",
        "entries": [
          {
            "confidence": "high",
            "count": 50,
            "count_basis": "questions across public and private splits",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-dbqa-vax-response-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Vaccine-response gene-set retrieval.",
            "reporting_status": "reported",
            "task_type_id": "scientific-database-retrieval"
          }
        ],
        "notes": "Single-purpose formal LAB-Bench track.",
        "status": "complete"
      },
      "summary": "Retrieves membership in MSigDB vaccine-response gene sets.",
      "task_counts": {
        "basis": "questions across public and private splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "questions",
            "count": 40,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-dbqa-vax-response-public",
            "label": "Public split",
            "notes": "Released in the official repository and Hugging Face dataset.",
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 10,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-dbqa-vax-response-private",
            "label": "Private contamination-monitoring split",
            "notes": "IDs are published; question contents are withheld.",
            "reporting_status": "reported"
          }
        ],
        "total": 50
      },
      "task_formats": [
        "multiple choice"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task count and public/private membership independently recomputed from the pinned split manifest.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2024-07-17",
          "evidence_ids": [
            "lab-bench-dbqa-vax-response-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "lab-bench-dbqa-vax-response-paper-v3",
          "label": "paper-v3",
          "notes": "Creator-paper full task snapshot used for Tables 2–4.",
          "release_date": "2024-07-17",
          "status": "active",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-vax-response-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-vax-response-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "lab-bench-dbqa-vax-response-evidence-repository"
          ],
          "formal_tracks": [],
          "id": "lab-bench-dbqa-vax-response-repository-998a8e0",
          "label": "repository-998a8e0",
          "notes": "Current commit-pinned task snapshot.",
          "release_date": "2025-09-27",
          "status": "current",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-vax-response-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-vax-response-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official repository publishes the public questions and both public/private ID manifests.",
        "biosafety_notes": "BioBench Atlas stores metadata and aggregate results only and does not mirror question or sequence content.",
        "grader": "The official harness scores exact single-choice outputs and reports accuracy, precision, coverage, and n.",
        "level": "partially-open",
        "license": "CC-BY-SA-4.0",
        "tasks": "40 questions are public and 10 question contents are withheld."
      },
      "aliases": [
        "DbQA — Viral protein–protein interactions"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": null,
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "retrieval",
        "prediction",
        "knowledge"
      ],
      "coverage_notes": [
        {
          "count": 50,
          "coverage": "explicitly-in-scope",
          "notes": "All questions query predicted viral–human protein interactions.",
          "reporting_status": "reported",
          "tag": "protein-protein-binding"
        }
      ],
      "domains": [
        "protein-protein-binding",
        "protein-science",
        "virology",
        "bioinformatics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-dbqa-viral-ppi-creator-mcq"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-viral-ppi-evidence-paper",
          "locator": {
            "note": "Task definition, total count, creator protocol, and result row.",
            "type": "section",
            "value": "Table 1; Appendix Tables 2–6; task description for DbQA_viral_ppi_task-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/kind",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0",
            "/scientific_task_classification/entries/1"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-viral-ppi-evidence-repository",
          "locator": {
            "note": "Public/private/total 40/10/50.",
            "type": "repository-path",
            "value": "DbQA/viral_ppi_task-v1-splits.json; DbQA/viral_ppi_task-v1-public.jsonl where present; task.py at 998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "source_id": "lab-bench-dbqa-viral-ppi-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets"
          ]
        }
      ],
      "field_status": [],
      "id": "lab-bench-dbqa-viral-ppi",
      "implementations": [
        {
          "commit": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838",
          "framework": "labbench Python evaluator",
          "notes": "Official task loader and exact-choice evaluator integration.",
          "status": "official",
          "url": "https://github.com/Future-House/LAB-Bench/blob/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/DbQA/task.py"
        }
      ],
      "kind": "track",
      "latest_version": "repository-998a8e0",
      "modalities": [
        "text",
        "database",
        "web"
      ],
      "name": "LAB-Bench DbQA — Viral protein–protein interactions",
      "organizations": [
        "FutureHouse"
      ],
      "parent_id": "lab-bench-dbqa",
      "release_date": "2024-07-14",
      "resources": [
        {
          "access_notes": "Commit-pinned official task files with prefix DbQA/viral_ppi_task-v1.",
          "id": "lab-bench-dbqa-viral-ppi-repository-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-SA-4.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/DbQA",
            "value": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "type": "repository",
          "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/DbQA"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "repository-998a8e0",
        "entries": [
          {
            "confidence": "high",
            "count": 50,
            "count_basis": "questions across public and private splits",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-dbqa-viral-ppi-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "The official task retrieves predicted P-HIPSter interaction partners.",
            "reporting_status": "reported",
            "task_type_id": "protein-protein-interaction-prediction"
          },
          {
            "confidence": "high",
            "count": 50,
            "count_basis": "questions across public and private splits",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-dbqa-viral-ppi-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Database retrieval is part of the evaluated task.",
            "reporting_status": "reported",
            "task_type_id": "scientific-database-retrieval"
          }
        ],
        "notes": "Formal viral-human protein-interaction retrieval track.",
        "status": "complete"
      },
      "summary": "Retrieves predicted human interaction partners of viral proteins from P-HIPSter.",
      "task_counts": {
        "basis": "questions across public and private splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "questions",
            "count": 40,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-dbqa-viral-ppi-public",
            "label": "Public split",
            "notes": "Released in the official repository and Hugging Face dataset.",
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 10,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-dbqa-viral-ppi-private",
            "label": "Private contamination-monitoring split",
            "notes": "IDs are published; question contents are withheld.",
            "reporting_status": "reported"
          }
        ],
        "total": 50
      },
      "task_formats": [
        "multiple choice"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task count and public/private membership independently recomputed from the pinned split manifest.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2024-07-17",
          "evidence_ids": [
            "lab-bench-dbqa-viral-ppi-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "lab-bench-dbqa-viral-ppi-paper-v3",
          "label": "paper-v3",
          "notes": "Creator-paper full task snapshot used for Tables 2–4.",
          "release_date": "2024-07-17",
          "status": "active",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-viral-ppi-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-viral-ppi-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "lab-bench-dbqa-viral-ppi-evidence-repository"
          ],
          "formal_tracks": [],
          "id": "lab-bench-dbqa-viral-ppi-repository-998a8e0",
          "label": "repository-998a8e0",
          "notes": "Current commit-pinned task snapshot.",
          "release_date": "2025-09-27",
          "status": "current",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-viral-ppi-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-dbqa-viral-ppi-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official repository publishes the public questions and both public/private ID manifests.",
        "biosafety_notes": "BioBench Atlas stores metadata and aggregate results only and does not mirror question or sequence content.",
        "grader": "The official harness scores exact single-choice outputs and reports accuracy, precision, coverage, and n.",
        "level": "partially-open",
        "license": "CC-BY-SA-4.0",
        "tasks": "181 questions are public and 45 question contents are withheld."
      },
      "aliases": [
        "FigQA"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": null,
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "evidence-synthesis",
        "scientific-reasoning"
      ],
      "coverage_notes": [],
      "domains": [
        "life-science",
        "molecular-cell-biology"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "anthropic-sonnet-4-5-system-card",
        "anthropic-sonnet-4-6-system-card",
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-figqa-anthropic-sonnet45-system-card",
        "lab-bench-figqa-creator-mcq",
        "lab-bench-figqa-creator-open-response",
        "lab-bench-figqa-crop-tool",
        "lab-bench-figqa-no-tools"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-figqa-evidence-paper",
          "locator": {
            "note": "Task definition, total count, creator protocol, and result row.",
            "type": "section",
            "value": "Table 1; Appendix Tables 2–6; task description for FigQA"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/kind",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-figqa-evidence-repository",
          "locator": {
            "note": "Public/private/total 181/45/226.",
            "type": "repository-path",
            "value": "FigQA/figqa-v1-splits.json; FigQA/figqa-v1-public.jsonl where present; task.py at 998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "source_id": "lab-bench-figqa-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets"
          ]
        }
      ],
      "field_status": [],
      "id": "lab-bench-figqa",
      "implementations": [
        {
          "commit": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838",
          "framework": "labbench Python evaluator",
          "notes": "Official task loader and exact-choice evaluator integration.",
          "status": "official",
          "url": "https://github.com/Future-House/LAB-Bench/blob/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/FigQA/task.py"
        }
      ],
      "kind": "track",
      "latest_version": "repository-998a8e0",
      "modalities": [
        "text",
        "figure",
        "image"
      ],
      "name": "LAB-Bench FigQA",
      "organizations": [
        "FutureHouse"
      ],
      "parent_id": "lab-bench",
      "release_date": "2024-07-14",
      "resources": [
        {
          "access_notes": "Commit-pinned official task files with prefix FigQA/figqa-v1.",
          "id": "lab-bench-figqa-repository-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-SA-4.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/FigQA",
            "value": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "type": "repository",
          "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/FigQA"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "repository-998a8e0",
        "entries": [
          {
            "confidence": "high",
            "count": 226,
            "count_basis": "questions across public and private splits",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-figqa-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Scientific figure interpretation.",
            "reporting_status": "reported",
            "task_type_id": "scientific-evidence-interpretation"
          }
        ],
        "notes": "Single-purpose formal LAB-Bench track.",
        "status": "complete"
      },
      "summary": "Multiple-choice interpretation and multi-element reasoning over scientific figures shown without captions or paper context.",
      "task_counts": {
        "basis": "questions across public and private splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "questions",
            "count": 181,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-figqa-public",
            "label": "Public split",
            "notes": "Released in the official repository and Hugging Face dataset.",
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 45,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-figqa-private",
            "label": "Private contamination-monitoring split",
            "notes": "IDs are published; question contents are withheld.",
            "reporting_status": "reported"
          },
          {
            "basis": "modified questions sampled for the creator open-response study",
            "count": 10,
            "exclusive": false,
            "exhaustive": false,
            "id": "lab-bench-figqa-creator-open-response",
            "label": "Creator open-response study",
            "notes": "A non-exhaustive derived subset; wording was modified to remove multiple-choice framing.",
            "reporting_status": "reported"
          }
        ],
        "total": 226
      },
      "task_formats": [
        "multiple choice"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task count and public/private membership independently recomputed from the pinned split manifest.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2024-07-17",
          "evidence_ids": [
            "lab-bench-figqa-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "lab-bench-figqa-paper-v3",
          "label": "paper-v3",
          "notes": "Creator-paper full task snapshot used for Tables 2–4.",
          "release_date": "2024-07-17",
          "status": "active",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 181,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-figqa-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 45,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-figqa-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              },
              {
                "basis": "modified questions sampled for the creator open-response study",
                "count": 10,
                "exclusive": false,
                "exhaustive": false,
                "id": "lab-bench-figqa-creator-open-response",
                "label": "Creator open-response study",
                "notes": "A non-exhaustive derived subset; wording was modified to remove multiple-choice framing.",
                "reporting_status": "reported"
              }
            ],
            "total": 226
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "lab-bench-figqa-evidence-repository"
          ],
          "formal_tracks": [],
          "id": "lab-bench-figqa-repository-998a8e0",
          "label": "repository-998a8e0",
          "notes": "Current commit-pinned task snapshot.",
          "release_date": "2025-09-27",
          "status": "current",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 181,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-figqa-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 45,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-figqa-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              },
              {
                "basis": "modified questions sampled for the creator open-response study",
                "count": 10,
                "exclusive": false,
                "exhaustive": false,
                "id": "lab-bench-figqa-creator-open-response",
                "label": "Creator open-response study",
                "notes": "A non-exhaustive derived subset; wording was modified to remove multiple-choice framing.",
                "reporting_status": "reported"
              }
            ],
            "total": 226
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official repository publishes the public questions and both public/private ID manifests.",
        "biosafety_notes": "BioBench Atlas stores metadata and aggregate results only and does not mirror question or sequence content.",
        "grader": "The official harness scores exact single-choice outputs and reports accuracy, precision, coverage, and n.",
        "level": "partially-open",
        "license": "CC-BY-SA-4.0",
        "tasks": "199 questions are public and 49 question contents are withheld."
      },
      "aliases": [
        "LitQA2"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": null,
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "retrieval",
        "evidence-synthesis",
        "scientific-reasoning"
      ],
      "coverage_notes": [],
      "domains": [
        "life-science"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-litqa2-creator-mcq"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-litqa2-evidence-paper",
          "locator": {
            "note": "Task definition, total count, creator protocol, and result row.",
            "type": "section",
            "value": "Table 1; Appendix Tables 2–6; task description for LitQA2"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/kind",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-litqa2-evidence-repository",
          "locator": {
            "note": "Public/private/total 199/49/248.",
            "type": "repository-path",
            "value": "LitQA2/litqa-v2-splits.json; LitQA2/litqa-v2-public.jsonl where present; task.py at 998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "source_id": "lab-bench-litqa2-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets"
          ]
        }
      ],
      "field_status": [],
      "id": "lab-bench-litqa2",
      "implementations": [
        {
          "commit": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838",
          "framework": "labbench Python evaluator",
          "notes": "Official task loader and exact-choice evaluator integration.",
          "status": "official",
          "url": "https://github.com/Future-House/LAB-Bench/blob/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/LitQA2/task.py"
        }
      ],
      "kind": "track",
      "latest_version": "repository-998a8e0",
      "modalities": [
        "text",
        "paper",
        "web"
      ],
      "name": "LAB-Bench LitQA2",
      "organizations": [
        "FutureHouse"
      ],
      "parent_id": "lab-bench",
      "release_date": "2024-07-14",
      "resources": [
        {
          "access_notes": "Commit-pinned official task files with prefix LitQA2/litqa-v2.",
          "id": "lab-bench-litqa2-repository-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-SA-4.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/LitQA2",
            "value": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "type": "repository",
          "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/LitQA2"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "repository-998a8e0",
        "entries": [
          {
            "confidence": "high",
            "count": 248,
            "count_basis": "questions across public and private splits",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-litqa2-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Full-paper evidence retrieval and interpretation.",
            "reporting_status": "reported",
            "task_type_id": "scientific-evidence-interpretation"
          }
        ],
        "notes": "Single-purpose formal LAB-Bench track.",
        "status": "complete"
      },
      "summary": "Literature-retrieval questions whose answers require findings in full research papers rather than titles or abstracts.",
      "task_counts": {
        "basis": "questions across public and private splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "questions",
            "count": 199,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-litqa2-public",
            "label": "Public split",
            "notes": "Released in the official repository and Hugging Face dataset.",
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 49,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-litqa2-private",
            "label": "Private contamination-monitoring split",
            "notes": "IDs are published; question contents are withheld.",
            "reporting_status": "reported"
          }
        ],
        "total": 248
      },
      "task_formats": [
        "multiple choice"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task count and public/private membership independently recomputed from the pinned split manifest.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2024-07-17",
          "evidence_ids": [
            "lab-bench-litqa2-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "lab-bench-litqa2-paper-v3",
          "label": "paper-v3",
          "notes": "Creator-paper full task snapshot used for Tables 2–4.",
          "release_date": "2024-07-17",
          "status": "active",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 199,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-litqa2-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 49,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-litqa2-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 248
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "lab-bench-litqa2-evidence-repository"
          ],
          "formal_tracks": [],
          "id": "lab-bench-litqa2-repository-998a8e0",
          "label": "repository-998a8e0",
          "notes": "Current commit-pinned task snapshot.",
          "release_date": "2025-09-27",
          "status": "current",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 199,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-litqa2-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 49,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-litqa2-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 248
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official repository publishes the public questions and both public/private ID manifests.",
        "biosafety_notes": "BioBench Atlas stores metadata and aggregate results only and does not mirror question or sequence content. Benchmark answers must not be treated as operational laboratory guidance.",
        "grader": "The official harness scores exact single-choice outputs and reports accuracy, precision, coverage, and n.",
        "level": "partially-open",
        "license": "CC-BY-SA-4.0",
        "tasks": "108 questions are public and 27 question contents are withheld."
      },
      "aliases": [
        "ProtocolQA"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": null,
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "troubleshooting",
        "scientific-reasoning"
      ],
      "coverage_notes": [],
      "domains": [
        "assay-screening",
        "molecular-cell-biology"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "anthropic-life-sciences",
        "anthropic-sonnet-4-5-system-card",
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-protocolqa-anthropic",
        "lab-bench-protocolqa-anthropic-sonnet45-system-card",
        "lab-bench-protocolqa-creator-mcq",
        "lab-bench-protocolqa-creator-open-response"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-protocolqa-evidence-paper",
          "locator": {
            "note": "Task definition, total count, creator protocol, and result row.",
            "type": "section",
            "value": "Table 1; Appendix Tables 2–6; task description for ProtocolQA"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/kind",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-protocolqa-evidence-repository",
          "locator": {
            "note": "Public/private/total 108/27/135.",
            "type": "repository-path",
            "value": "ProtocolQA/protocolqa-v1-splits.json; ProtocolQA/protocolqa-v1-public.jsonl where present; task.py at 998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "source_id": "lab-bench-protocolqa-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets"
          ]
        }
      ],
      "field_status": [],
      "id": "lab-bench-protocolqa",
      "implementations": [
        {
          "commit": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838",
          "framework": "labbench Python evaluator",
          "notes": "Official task loader and exact-choice evaluator integration.",
          "status": "official",
          "url": "https://github.com/Future-House/LAB-Bench/blob/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/ProtocolQA/task.py"
        }
      ],
      "kind": "track",
      "latest_version": "repository-998a8e0",
      "modalities": [
        "text"
      ],
      "name": "LAB-Bench ProtocolQA",
      "organizations": [
        "FutureHouse"
      ],
      "parent_id": "lab-bench",
      "release_date": "2024-07-14",
      "resources": [
        {
          "access_notes": "Commit-pinned official task files with prefix ProtocolQA/protocolqa-v1.",
          "id": "lab-bench-protocolqa-repository-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-SA-4.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/ProtocolQA",
            "value": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "type": "repository",
          "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/ProtocolQA"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "repository-998a8e0",
        "entries": [
          {
            "confidence": "high",
            "count": 135,
            "count_basis": "questions across public and private splits",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-protocolqa-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Protocol troubleshooting is classified under protocol planning.",
            "reporting_status": "reported",
            "task_type_id": "experiment-protocol-planning"
          }
        ],
        "notes": "Single-purpose formal LAB-Bench track.",
        "status": "complete"
      },
      "summary": "Troubleshoots intentionally modified published biological protocols by selecting steps that would repair the stated outcome.",
      "task_counts": {
        "basis": "questions across public and private splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "questions",
            "count": 108,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-protocolqa-public",
            "label": "Public split",
            "notes": "Released in the official repository and Hugging Face dataset.",
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 27,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-protocolqa-private",
            "label": "Private contamination-monitoring split",
            "notes": "IDs are published; question contents are withheld.",
            "reporting_status": "reported"
          },
          {
            "basis": "modified questions sampled for the creator open-response study",
            "count": 20,
            "exclusive": false,
            "exhaustive": false,
            "id": "lab-bench-protocolqa-creator-open-response",
            "label": "Creator open-response study",
            "notes": "A non-exhaustive derived subset; wording was modified to remove multiple-choice framing.",
            "reporting_status": "reported"
          }
        ],
        "total": 135
      },
      "task_formats": [
        "multiple choice"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task count and public/private membership independently recomputed from the pinned split manifest.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2024-07-17",
          "evidence_ids": [
            "lab-bench-protocolqa-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "lab-bench-protocolqa-paper-v3",
          "label": "paper-v3",
          "notes": "Creator-paper full task snapshot used for Tables 2–4.",
          "release_date": "2024-07-17",
          "status": "active",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 108,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-protocolqa-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 27,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-protocolqa-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              },
              {
                "basis": "modified questions sampled for the creator open-response study",
                "count": 20,
                "exclusive": false,
                "exhaustive": false,
                "id": "lab-bench-protocolqa-creator-open-response",
                "label": "Creator open-response study",
                "notes": "A non-exhaustive derived subset; wording was modified to remove multiple-choice framing.",
                "reporting_status": "reported"
              }
            ],
            "total": 135
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "lab-bench-protocolqa-evidence-repository"
          ],
          "formal_tracks": [],
          "id": "lab-bench-protocolqa-repository-998a8e0",
          "label": "repository-998a8e0",
          "notes": "Current commit-pinned task snapshot.",
          "release_date": "2025-09-27",
          "status": "current",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 108,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-protocolqa-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 27,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-protocolqa-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              },
              {
                "basis": "modified questions sampled for the creator open-response study",
                "count": 20,
                "exclusive": false,
                "exhaustive": false,
                "id": "lab-bench-protocolqa-creator-open-response",
                "label": "Creator open-response study",
                "notes": "A non-exhaustive derived subset; wording was modified to remove multiple-choice framing.",
                "reporting_status": "reported"
              }
            ],
            "total": 135
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "Public questions and all public/private split IDs are versioned in the official repository.",
        "biosafety_notes": "BioBench Atlas stores metadata and aggregate results only.",
        "grader": "The official harness scores exact single-choice outputs and reports accuracy, precision, coverage, and n.",
        "level": "partially-open",
        "license": "CC-BY-SA-4.0",
        "tasks": "600 questions are public and 150 question contents are withheld."
      },
      "aliases": [
        "SeqQA"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": null,
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "data-analysis",
        "design",
        "prediction",
        "scientific-reasoning",
        "tool-use"
      ],
      "coverage_notes": [
        {
          "count": null,
          "coverage": "explicitly-in-scope",
          "notes": "ORF tasks connect DNA/RNA and amino-acid outputs, but no official protein-only question count is published.",
          "reporting_status": "not_reported",
          "tag": "protein-sequence"
        },
        {
          "count": 0,
          "coverage": "not-in-scope",
          "notes": "Primer design is included; protein sequence design is not.",
          "reporting_status": "reported",
          "tag": "protein-design"
        }
      ],
      "domains": [
        "genomics",
        "transcriptomics",
        "protein-sequence",
        "molecular-cell-biology"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "anthropic-sonnet-4-5-system-card"
      ],
      "evaluation_run_ids": [
        "lab-bench-seqqa-anthropic-sonnet45-system-card"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-evidence-paper",
          "locator": {
            "note": "Category total, task definitions, domains, modalities, and creator evaluation.",
            "type": "table",
            "value": "Table 1 and Appendix Table 6; task-specific Appendix C sections"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/kind",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/license",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0",
            "/scientific_task_classification/entries/1"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-evidence-repository",
          "locator": {
            "note": "Public/private/total 600/150/750; exact child files and evaluator.",
            "type": "repository-path",
            "value": "SeqQA/*-splits.json and task.py at 998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "source_id": "lab-bench-seqqa-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets"
          ]
        }
      ],
      "field_status": [],
      "id": "lab-bench-seqqa",
      "implementations": [
        {
          "commit": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838",
          "framework": "labbench Python evaluator",
          "notes": "Official category loader and evaluator.",
          "status": "official",
          "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
        }
      ],
      "kind": "track",
      "latest_version": "repository-998a8e0",
      "modalities": [
        "text",
        "dna-rna-sequence",
        "protein-sequence"
      ],
      "name": "LAB-Bench SeqQA",
      "organizations": [
        "FutureHouse"
      ],
      "parent_id": "lab-bench",
      "release_date": "2024-07-14",
      "resources": [
        {
          "access_notes": "Commit-pinned official category directory.",
          "id": "lab-bench-seqqa-repository-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-SA-4.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA",
            "value": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "type": "repository",
          "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "repository-998a8e0",
        "entries": [
          {
            "confidence": "high",
            "count": null,
            "count_basis": "Primer-design questions across formal SeqQA child tracks.",
            "count_ref": null,
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-seqqa-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "A cross-track subtotal is not reported by the creator.",
            "reporting_status": "not_reported",
            "task_type_id": "dna-sequence-design"
          },
          {
            "confidence": "high",
            "count": 200,
            "count_basis": "Four 50-question ORF and translation formal child tracks.",
            "count_ref": null,
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-seqqa-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "The count is derived only from formal child-track sizes documented by the official release.",
            "reporting_status": "reported",
            "task_type_id": "rna-processing-translation-prediction"
          }
        ],
        "notes": "Formal child tracks are mapped where they express a controlled Scientific Task; generic sequence arithmetic remains unclassified.",
        "status": "partial"
      },
      "summary": "Sequence-comprehension and manipulation category spanning 15 formal tasks involving PCR, restriction digestion, ORFs, translation, GC content, and DNA–protein relationships.",
      "task_counts": {
        "basis": "questions across formal child tasks",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "questions",
            "count": 50,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-orf-seq-aaid",
            "label": "ORF amino-acid position",
            "notes": null,
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 50,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-orf-seq-aaseq",
            "label": "ORF amino-acid sequence",
            "notes": null,
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 50,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-orf-seq-numlen",
            "label": "ORF count above length",
            "notes": null,
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 50,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-orf-transeff",
            "label": "Translation efficiency",
            "notes": null,
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 50,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-pcr-gene-enzprimers",
            "label": "Gene-to-restriction primers",
            "notes": null,
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 50,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-pcr-gene-gibshindprimers",
            "label": "Gene-to-Gibson primers (HindIII)",
            "notes": null,
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 50,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-pcr-gene-gibssmaprimers",
            "label": "Gene-to-Gibson primers (SmaI)",
            "notes": null,
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 50,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-pcr-geneprimers-enz",
            "label": "Primers-to-restriction enzymes",
            "notes": null,
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 50,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-pcr-len-primers",
            "label": "Amplicon length to primers",
            "notes": null,
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 50,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-pcr-primers-len",
            "label": "Primers to amplicon length",
            "notes": null,
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 50,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-pcr-seq-enzprimers",
            "label": "Sequence-to-restriction primers",
            "notes": null,
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 50,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-pcr-seq-primers",
            "label": "Amplicon sequence to primers",
            "notes": null,
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 50,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-prop-seq-gcpercent",
            "label": "GC percentage",
            "notes": null,
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 50,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-re-seq-lenfrags",
            "label": "Restriction-fragment lengths",
            "notes": null,
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 50,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-re-seq-numfrags",
            "label": "Restriction-fragment count",
            "notes": null,
            "reporting_status": "reported"
          }
        ],
        "total": 750
      },
      "task_formats": [
        "multiple choice"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Counts and child-task relationships independently recomputed from versioned split manifests.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2024-07-17",
          "evidence_ids": [
            "lab-bench-seqqa-evidence-paper"
          ],
          "formal_tracks": [
            "lab-bench-seqqa-orf-seq-aaid",
            "lab-bench-seqqa-orf-seq-aaseq",
            "lab-bench-seqqa-orf-seq-numlen",
            "lab-bench-seqqa-orf-transeff",
            "lab-bench-seqqa-pcr-gene-enzprimers",
            "lab-bench-seqqa-pcr-gene-gibshindprimers",
            "lab-bench-seqqa-pcr-gene-gibssmaprimers",
            "lab-bench-seqqa-pcr-geneprimers-enz",
            "lab-bench-seqqa-pcr-len-primers",
            "lab-bench-seqqa-pcr-primers-len",
            "lab-bench-seqqa-pcr-seq-enzprimers",
            "lab-bench-seqqa-pcr-seq-primers",
            "lab-bench-seqqa-prop-seq-gcpercent",
            "lab-bench-seqqa-re-seq-lenfrags",
            "lab-bench-seqqa-re-seq-numfrags"
          ],
          "id": "lab-bench-seqqa-paper-v3",
          "label": "paper-v3",
          "notes": "Creator-paper full category snapshot.",
          "release_date": "2024-07-17",
          "status": "active",
          "task_counts": {
            "basis": "questions across formal child tasks",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-orf-seq-aaid",
                "label": "ORF amino-acid position",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-orf-seq-aaseq",
                "label": "ORF amino-acid sequence",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-orf-seq-numlen",
                "label": "ORF count above length",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-orf-transeff",
                "label": "Translation efficiency",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-gene-enzprimers",
                "label": "Gene-to-restriction primers",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-gene-gibshindprimers",
                "label": "Gene-to-Gibson primers (HindIII)",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-gene-gibssmaprimers",
                "label": "Gene-to-Gibson primers (SmaI)",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-geneprimers-enz",
                "label": "Primers-to-restriction enzymes",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-len-primers",
                "label": "Amplicon length to primers",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-primers-len",
                "label": "Primers to amplicon length",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-seq-enzprimers",
                "label": "Sequence-to-restriction primers",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-seq-primers",
                "label": "Amplicon sequence to primers",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-prop-seq-gcpercent",
                "label": "GC percentage",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-re-seq-lenfrags",
                "label": "Restriction-fragment lengths",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-re-seq-numfrags",
                "label": "Restriction-fragment count",
                "notes": null,
                "reporting_status": "reported"
              }
            ],
            "total": 750
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "lab-bench-seqqa-evidence-repository"
          ],
          "formal_tracks": [
            "lab-bench-seqqa-orf-seq-aaid",
            "lab-bench-seqqa-orf-seq-aaseq",
            "lab-bench-seqqa-orf-seq-numlen",
            "lab-bench-seqqa-orf-transeff",
            "lab-bench-seqqa-pcr-gene-enzprimers",
            "lab-bench-seqqa-pcr-gene-gibshindprimers",
            "lab-bench-seqqa-pcr-gene-gibssmaprimers",
            "lab-bench-seqqa-pcr-geneprimers-enz",
            "lab-bench-seqqa-pcr-len-primers",
            "lab-bench-seqqa-pcr-primers-len",
            "lab-bench-seqqa-pcr-seq-enzprimers",
            "lab-bench-seqqa-pcr-seq-primers",
            "lab-bench-seqqa-prop-seq-gcpercent",
            "lab-bench-seqqa-re-seq-lenfrags",
            "lab-bench-seqqa-re-seq-numfrags"
          ],
          "id": "lab-bench-seqqa-repository-998a8e0",
          "label": "repository-998a8e0",
          "notes": "Current commit-pinned category snapshot.",
          "release_date": "2025-09-27",
          "status": "current",
          "task_counts": {
            "basis": "questions across formal child tasks",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-orf-seq-aaid",
                "label": "ORF amino-acid position",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-orf-seq-aaseq",
                "label": "ORF amino-acid sequence",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-orf-seq-numlen",
                "label": "ORF count above length",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-orf-transeff",
                "label": "Translation efficiency",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-gene-enzprimers",
                "label": "Gene-to-restriction primers",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-gene-gibshindprimers",
                "label": "Gene-to-Gibson primers (HindIII)",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-gene-gibssmaprimers",
                "label": "Gene-to-Gibson primers (SmaI)",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-geneprimers-enz",
                "label": "Primers-to-restriction enzymes",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-len-primers",
                "label": "Amplicon length to primers",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-primers-len",
                "label": "Primers to amplicon length",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-seq-enzprimers",
                "label": "Sequence-to-restriction primers",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-seq-primers",
                "label": "Amplicon sequence to primers",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-prop-seq-gcpercent",
                "label": "GC percentage",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-re-seq-lenfrags",
                "label": "Restriction-fragment lengths",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 50,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-re-seq-numfrags",
                "label": "Restriction-fragment count",
                "notes": null,
                "reporting_status": "reported"
              }
            ],
            "total": 750
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official repository publishes the public questions and both public/private ID manifests.",
        "biosafety_notes": "BioBench Atlas stores metadata and aggregate results only and does not mirror question or sequence content.",
        "grader": "The official harness scores exact single-choice outputs and reports accuracy, precision, coverage, and n.",
        "level": "partially-open",
        "license": "CC-BY-SA-4.0",
        "tasks": "40 questions are public and 10 question contents are withheld."
      },
      "aliases": [
        "ORF-seq-AApos"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": null,
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "data-analysis",
        "scientific-reasoning"
      ],
      "coverage_notes": [
        {
          "count": null,
          "coverage": "explicitly-in-scope",
          "notes": "The task links nucleotide ORFs and encoded proteins; protein-input/output-only counts are not separately reported.",
          "reporting_status": "not_reported",
          "tag": "protein-sequence"
        }
      ],
      "domains": [
        "genomics",
        "protein-sequence"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-seqqa-orf-seq-aaid-creator-mcq"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-orf-seq-aaid-evidence-paper",
          "locator": {
            "note": "Task definition, total count, creator protocol, and result row.",
            "type": "section",
            "value": "Table 1; Appendix Tables 2–6; task description for SeqQA_ORF-seq-AAid-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/kind",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-orf-seq-aaid-evidence-repository",
          "locator": {
            "note": "Public/private/total 40/10/50.",
            "type": "repository-path",
            "value": "SeqQA/ORF-seq-AAid-v1-splits.json; SeqQA/ORF-seq-AAid-v1-public.jsonl where present; task.py at 998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "source_id": "lab-bench-seqqa-orf-seq-aaid-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets"
          ]
        }
      ],
      "field_status": [],
      "id": "lab-bench-seqqa-orf-seq-aaid",
      "implementations": [
        {
          "commit": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838",
          "framework": "labbench Python evaluator",
          "notes": "Official task loader and exact-choice evaluator integration.",
          "status": "official",
          "url": "https://github.com/Future-House/LAB-Bench/blob/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA/task.py"
        }
      ],
      "kind": "track",
      "latest_version": "repository-998a8e0",
      "modalities": [
        "text",
        "dna-rna-sequence",
        "protein-sequence"
      ],
      "name": "LAB-Bench SeqQA — ORF amino-acid position",
      "organizations": [
        "FutureHouse"
      ],
      "parent_id": "lab-bench-seqqa",
      "release_date": "2024-07-14",
      "resources": [
        {
          "access_notes": "Commit-pinned official task files with prefix SeqQA/ORF-seq-AAid-v1.",
          "id": "lab-bench-seqqa-orf-seq-aaid-repository-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-SA-4.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA",
            "value": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "type": "repository",
          "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "repository-998a8e0",
        "entries": [
          {
            "confidence": "high",
            "count": 50,
            "count_basis": "questions across public and private splits",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-seqqa-orf-seq-aaid-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Translation of a DNA open reading frame.",
            "reporting_status": "reported",
            "task_type_id": "rna-processing-translation-prediction"
          }
        ],
        "notes": "Single-purpose formal LAB-Bench track.",
        "status": "complete"
      },
      "summary": "Finds the amino acid encoded at a specified position in the longest ORF of a DNA sequence.",
      "task_counts": {
        "basis": "questions across public and private splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "questions",
            "count": 40,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-orf-seq-aaid-public",
            "label": "Public split",
            "notes": "Released in the official repository and Hugging Face dataset.",
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 10,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-orf-seq-aaid-private",
            "label": "Private contamination-monitoring split",
            "notes": "IDs are published; question contents are withheld.",
            "reporting_status": "reported"
          }
        ],
        "total": 50
      },
      "task_formats": [
        "multiple choice"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task count and public/private membership independently recomputed from the pinned split manifest.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2024-07-17",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaid-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "lab-bench-seqqa-orf-seq-aaid-paper-v3",
          "label": "paper-v3",
          "notes": "Creator-paper full task snapshot used for Tables 2–4.",
          "release_date": "2024-07-17",
          "status": "active",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-orf-seq-aaid-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-orf-seq-aaid-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaid-evidence-repository"
          ],
          "formal_tracks": [],
          "id": "lab-bench-seqqa-orf-seq-aaid-repository-998a8e0",
          "label": "repository-998a8e0",
          "notes": "Current commit-pinned task snapshot.",
          "release_date": "2025-09-27",
          "status": "current",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-orf-seq-aaid-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-orf-seq-aaid-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official repository publishes the public questions and both public/private ID manifests.",
        "biosafety_notes": "BioBench Atlas stores metadata and aggregate results only and does not mirror question or sequence content.",
        "grader": "The official harness scores exact single-choice outputs and reports accuracy, precision, coverage, and n.",
        "level": "partially-open",
        "license": "CC-BY-SA-4.0",
        "tasks": "40 questions are public and 10 question contents are withheld."
      },
      "aliases": [
        "SeqQA — ORF amino-acid sequence"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": null,
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "data-analysis",
        "scientific-reasoning"
      ],
      "coverage_notes": [
        {
          "count": null,
          "coverage": "explicitly-in-scope",
          "notes": "The task links nucleotide ORFs and encoded proteins; protein-input/output-only counts are not separately reported.",
          "reporting_status": "not_reported",
          "tag": "protein-sequence"
        }
      ],
      "domains": [
        "genomics",
        "protein-sequence"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-seqqa-orf-seq-aaseq-creator-mcq"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-orf-seq-aaseq-evidence-paper",
          "locator": {
            "note": "Task definition, total count, creator protocol, and result row.",
            "type": "section",
            "value": "Table 1; Appendix Tables 2–6; task description for SeqQA_ORF-seq-AAseq-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/kind",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-orf-seq-aaseq-evidence-repository",
          "locator": {
            "note": "Public/private/total 40/10/50.",
            "type": "repository-path",
            "value": "SeqQA/ORF-seq-AAseq-v1-splits.json; SeqQA/ORF-seq-AAseq-v1-public.jsonl where present; task.py at 998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "source_id": "lab-bench-seqqa-orf-seq-aaseq-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets"
          ]
        }
      ],
      "field_status": [],
      "id": "lab-bench-seqqa-orf-seq-aaseq",
      "implementations": [
        {
          "commit": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838",
          "framework": "labbench Python evaluator",
          "notes": "Official task loader and exact-choice evaluator integration.",
          "status": "official",
          "url": "https://github.com/Future-House/LAB-Bench/blob/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA/task.py"
        }
      ],
      "kind": "track",
      "latest_version": "repository-998a8e0",
      "modalities": [
        "text",
        "dna-rna-sequence",
        "protein-sequence"
      ],
      "name": "LAB-Bench SeqQA — ORF amino-acid sequence",
      "organizations": [
        "FutureHouse"
      ],
      "parent_id": "lab-bench-seqqa",
      "release_date": "2024-07-14",
      "resources": [
        {
          "access_notes": "Commit-pinned official task files with prefix SeqQA/ORF-seq-AAseq-v1.",
          "id": "lab-bench-seqqa-orf-seq-aaseq-repository-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-SA-4.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA",
            "value": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "type": "repository",
          "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "repository-998a8e0",
        "entries": [
          {
            "confidence": "high",
            "count": 50,
            "count_basis": "questions across public and private splits",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-seqqa-orf-seq-aaseq-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Translation of a DNA open reading frame.",
            "reporting_status": "reported",
            "task_type_id": "rna-processing-translation-prediction"
          }
        ],
        "notes": "Single-purpose formal LAB-Bench track.",
        "status": "complete"
      },
      "summary": "Translates the longest ORF in a DNA sequence to its amino-acid sequence.",
      "task_counts": {
        "basis": "questions across public and private splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "questions",
            "count": 40,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-orf-seq-aaseq-public",
            "label": "Public split",
            "notes": "Released in the official repository and Hugging Face dataset.",
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 10,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-orf-seq-aaseq-private",
            "label": "Private contamination-monitoring split",
            "notes": "IDs are published; question contents are withheld.",
            "reporting_status": "reported"
          }
        ],
        "total": 50
      },
      "task_formats": [
        "multiple choice"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task count and public/private membership independently recomputed from the pinned split manifest.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2024-07-17",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaseq-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "lab-bench-seqqa-orf-seq-aaseq-paper-v3",
          "label": "paper-v3",
          "notes": "Creator-paper full task snapshot used for Tables 2–4.",
          "release_date": "2024-07-17",
          "status": "active",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-orf-seq-aaseq-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-orf-seq-aaseq-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaseq-evidence-repository"
          ],
          "formal_tracks": [],
          "id": "lab-bench-seqqa-orf-seq-aaseq-repository-998a8e0",
          "label": "repository-998a8e0",
          "notes": "Current commit-pinned task snapshot.",
          "release_date": "2025-09-27",
          "status": "current",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-orf-seq-aaseq-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-orf-seq-aaseq-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official repository publishes the public questions and both public/private ID manifests.",
        "biosafety_notes": "BioBench Atlas stores metadata and aggregate results only and does not mirror question or sequence content.",
        "grader": "The official harness scores exact single-choice outputs and reports accuracy, precision, coverage, and n.",
        "level": "partially-open",
        "license": "CC-BY-SA-4.0",
        "tasks": "40 questions are public and 10 question contents are withheld."
      },
      "aliases": [
        "SeqQA — ORF count above length"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": null,
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "data-analysis",
        "scientific-reasoning"
      ],
      "coverage_notes": [
        {
          "count": null,
          "coverage": "explicitly-in-scope",
          "notes": "The task links nucleotide ORFs and encoded proteins; protein-input/output-only counts are not separately reported.",
          "reporting_status": "not_reported",
          "tag": "protein-sequence"
        }
      ],
      "domains": [
        "genomics",
        "protein-sequence"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-seqqa-orf-seq-numlen-creator-mcq"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-orf-seq-numlen-evidence-paper",
          "locator": {
            "note": "Task definition, total count, creator protocol, and result row.",
            "type": "section",
            "value": "Table 1; Appendix Tables 2–6; task description for SeqQA_ORF-seq-numlen-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/kind",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-orf-seq-numlen-evidence-repository",
          "locator": {
            "note": "Public/private/total 40/10/50.",
            "type": "repository-path",
            "value": "SeqQA/ORF-seq-numlen-v1-splits.json; SeqQA/ORF-seq-numlen-v1-public.jsonl where present; task.py at 998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "source_id": "lab-bench-seqqa-orf-seq-numlen-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets"
          ]
        }
      ],
      "field_status": [],
      "id": "lab-bench-seqqa-orf-seq-numlen",
      "implementations": [
        {
          "commit": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838",
          "framework": "labbench Python evaluator",
          "notes": "Official task loader and exact-choice evaluator integration.",
          "status": "official",
          "url": "https://github.com/Future-House/LAB-Bench/blob/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA/task.py"
        }
      ],
      "kind": "track",
      "latest_version": "repository-998a8e0",
      "modalities": [
        "text",
        "dna-rna-sequence"
      ],
      "name": "LAB-Bench SeqQA — ORF count above length",
      "organizations": [
        "FutureHouse"
      ],
      "parent_id": "lab-bench-seqqa",
      "release_date": "2024-07-14",
      "resources": [
        {
          "access_notes": "Commit-pinned official task files with prefix SeqQA/ORF-seq-numlen-v1.",
          "id": "lab-bench-seqqa-orf-seq-numlen-repository-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-SA-4.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA",
            "value": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "type": "repository",
          "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "repository-998a8e0",
        "entries": [
          {
            "confidence": "high",
            "count": 50,
            "count_basis": "questions across public and private splits",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-seqqa-orf-seq-numlen-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Open-reading-frame interpretation.",
            "reporting_status": "reported",
            "task_type_id": "rna-processing-translation-prediction"
          }
        ],
        "notes": "Single-purpose formal LAB-Bench track.",
        "status": "complete"
      },
      "summary": "Counts open reading frames encoding proteins above a specified amino-acid length.",
      "task_counts": {
        "basis": "questions across public and private splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "questions",
            "count": 40,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-orf-seq-numlen-public",
            "label": "Public split",
            "notes": "Released in the official repository and Hugging Face dataset.",
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 10,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-orf-seq-numlen-private",
            "label": "Private contamination-monitoring split",
            "notes": "IDs are published; question contents are withheld.",
            "reporting_status": "reported"
          }
        ],
        "total": 50
      },
      "task_formats": [
        "multiple choice"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task count and public/private membership independently recomputed from the pinned split manifest.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2024-07-17",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-numlen-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "lab-bench-seqqa-orf-seq-numlen-paper-v3",
          "label": "paper-v3",
          "notes": "Creator-paper full task snapshot used for Tables 2–4.",
          "release_date": "2024-07-17",
          "status": "active",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-orf-seq-numlen-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-orf-seq-numlen-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-numlen-evidence-repository"
          ],
          "formal_tracks": [],
          "id": "lab-bench-seqqa-orf-seq-numlen-repository-998a8e0",
          "label": "repository-998a8e0",
          "notes": "Current commit-pinned task snapshot.",
          "release_date": "2025-09-27",
          "status": "current",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-orf-seq-numlen-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-orf-seq-numlen-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official repository publishes the public questions and both public/private ID manifests.",
        "biosafety_notes": "BioBench Atlas stores metadata and aggregate results only and does not mirror question or sequence content.",
        "grader": "The official harness scores exact single-choice outputs and reports accuracy, precision, coverage, and n.",
        "level": "partially-open",
        "license": "CC-BY-SA-4.0",
        "tasks": "40 questions are public and 10 question contents are withheld."
      },
      "aliases": [
        "ORF-optcodon"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": null,
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction",
        "scientific-reasoning"
      ],
      "coverage_notes": [
        {
          "count": null,
          "coverage": "explicitly-in-scope",
          "notes": "The task links nucleotide ORFs and encoded proteins; protein-input/output-only counts are not separately reported.",
          "reporting_status": "not_reported",
          "tag": "protein-sequence"
        }
      ],
      "domains": [
        "transcriptomics",
        "protein-sequence"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-seqqa-orf-transeff-creator-mcq"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-orf-transeff-evidence-paper",
          "locator": {
            "note": "Task definition, total count, creator protocol, and result row.",
            "type": "section",
            "value": "Table 1; Appendix Tables 2–6; task description for SeqQA_ORF-transeff-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/kind",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-orf-transeff-evidence-repository",
          "locator": {
            "note": "Public/private/total 40/10/50.",
            "type": "repository-path",
            "value": "SeqQA/ORF-transeff-v1-splits.json; SeqQA/ORF-transeff-v1-public.jsonl where present; task.py at 998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "source_id": "lab-bench-seqqa-orf-transeff-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets"
          ]
        }
      ],
      "field_status": [],
      "id": "lab-bench-seqqa-orf-transeff",
      "implementations": [
        {
          "commit": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838",
          "framework": "labbench Python evaluator",
          "notes": "Official task loader and exact-choice evaluator integration.",
          "status": "official",
          "url": "https://github.com/Future-House/LAB-Bench/blob/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA/task.py"
        }
      ],
      "kind": "track",
      "latest_version": "repository-998a8e0",
      "modalities": [
        "text",
        "dna-rna-sequence"
      ],
      "name": "LAB-Bench SeqQA — Translation efficiency",
      "organizations": [
        "FutureHouse"
      ],
      "parent_id": "lab-bench-seqqa",
      "release_date": "2024-07-14",
      "resources": [
        {
          "access_notes": "Commit-pinned official task files with prefix SeqQA/ORF-transeff-v1.",
          "id": "lab-bench-seqqa-orf-transeff-repository-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-SA-4.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA",
            "value": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "type": "repository",
          "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "repository-998a8e0",
        "entries": [
          {
            "confidence": "high",
            "count": 50,
            "count_basis": "questions across public and private splits",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-seqqa-orf-transeff-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Translation-efficiency reasoning.",
            "reporting_status": "reported",
            "task_type_id": "rna-processing-translation-prediction"
          }
        ],
        "notes": "Single-purpose formal LAB-Bench track.",
        "status": "complete"
      },
      "summary": "Selects an RNA sequence whose ORF context is most likely to yield high translation efficiency.",
      "task_counts": {
        "basis": "questions across public and private splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "questions",
            "count": 40,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-orf-transeff-public",
            "label": "Public split",
            "notes": "Released in the official repository and Hugging Face dataset.",
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 10,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-orf-transeff-private",
            "label": "Private contamination-monitoring split",
            "notes": "IDs are published; question contents are withheld.",
            "reporting_status": "reported"
          }
        ],
        "total": 50
      },
      "task_formats": [
        "multiple choice"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task count and public/private membership independently recomputed from the pinned split manifest.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2024-07-17",
          "evidence_ids": [
            "lab-bench-seqqa-orf-transeff-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "lab-bench-seqqa-orf-transeff-paper-v3",
          "label": "paper-v3",
          "notes": "Creator-paper full task snapshot used for Tables 2–4.",
          "release_date": "2024-07-17",
          "status": "active",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-orf-transeff-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-orf-transeff-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "lab-bench-seqqa-orf-transeff-evidence-repository"
          ],
          "formal_tracks": [],
          "id": "lab-bench-seqqa-orf-transeff-repository-998a8e0",
          "label": "repository-998a8e0",
          "notes": "Current commit-pinned task snapshot.",
          "release_date": "2025-09-27",
          "status": "current",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-orf-transeff-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-orf-transeff-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official repository publishes the public questions and both public/private ID manifests.",
        "biosafety_notes": "BioBench Atlas stores metadata and aggregate results only and does not mirror question or sequence content.",
        "grader": "The official harness scores exact single-choice outputs and reports accuracy, precision, coverage, and n.",
        "level": "partially-open",
        "license": "CC-BY-SA-4.0",
        "tasks": "40 questions are public and 10 question contents are withheld."
      },
      "aliases": [
        "SeqQA — Gene-to-restriction primers"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": null,
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "design",
        "scientific-reasoning"
      ],
      "coverage_notes": [],
      "domains": [
        "genomics",
        "molecular-cell-biology"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-seqqa-pcr-gene-enzprimers-creator-mcq"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-pcr-gene-enzprimers-evidence-paper",
          "locator": {
            "note": "Task definition, total count, creator protocol, and result row.",
            "type": "section",
            "value": "Table 1; Appendix Tables 2–6; task description for SeqQA_PCR-gene-enzprimers-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/kind",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-pcr-gene-enzprimers-evidence-repository",
          "locator": {
            "note": "Public/private/total 40/10/50.",
            "type": "repository-path",
            "value": "SeqQA/PCR-gene-enzprimers-v1-splits.json; SeqQA/PCR-gene-enzprimers-v1-public.jsonl where present; task.py at 998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "source_id": "lab-bench-seqqa-pcr-gene-enzprimers-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets"
          ]
        }
      ],
      "field_status": [],
      "id": "lab-bench-seqqa-pcr-gene-enzprimers",
      "implementations": [
        {
          "commit": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838",
          "framework": "labbench Python evaluator",
          "notes": "Official task loader and exact-choice evaluator integration.",
          "status": "official",
          "url": "https://github.com/Future-House/LAB-Bench/blob/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA/task.py"
        }
      ],
      "kind": "track",
      "latest_version": "repository-998a8e0",
      "modalities": [
        "text",
        "dna-rna-sequence"
      ],
      "name": "LAB-Bench SeqQA — Gene-to-restriction primers",
      "organizations": [
        "FutureHouse"
      ],
      "parent_id": "lab-bench-seqqa",
      "release_date": "2024-07-14",
      "resources": [
        {
          "access_notes": "Commit-pinned official task files with prefix SeqQA/PCR-gene-enzprimers-v1.",
          "id": "lab-bench-seqqa-pcr-gene-enzprimers-repository-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-SA-4.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA",
            "value": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "type": "repository",
          "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "repository-998a8e0",
        "entries": [
          {
            "confidence": "high",
            "count": 50,
            "count_basis": "questions across public and private splits",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-seqqa-pcr-gene-enzprimers-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Restriction-cloning primer design.",
            "reporting_status": "reported",
            "task_type_id": "dna-sequence-design"
          }
        ],
        "notes": "Single-purpose formal LAB-Bench track.",
        "status": "complete"
      },
      "summary": "Selects restriction-cloning primers from a named gene and enzyme pair.",
      "task_counts": {
        "basis": "questions across public and private splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "questions",
            "count": 40,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-pcr-gene-enzprimers-public",
            "label": "Public split",
            "notes": "Released in the official repository and Hugging Face dataset.",
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 10,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-pcr-gene-enzprimers-private",
            "label": "Private contamination-monitoring split",
            "notes": "IDs are published; question contents are withheld.",
            "reporting_status": "reported"
          }
        ],
        "total": 50
      },
      "task_formats": [
        "multiple choice"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task count and public/private membership independently recomputed from the pinned split manifest.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2024-07-17",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-enzprimers-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "lab-bench-seqqa-pcr-gene-enzprimers-paper-v3",
          "label": "paper-v3",
          "notes": "Creator-paper full task snapshot used for Tables 2–4.",
          "release_date": "2024-07-17",
          "status": "active",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-gene-enzprimers-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-gene-enzprimers-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-enzprimers-evidence-repository"
          ],
          "formal_tracks": [],
          "id": "lab-bench-seqqa-pcr-gene-enzprimers-repository-998a8e0",
          "label": "repository-998a8e0",
          "notes": "Current commit-pinned task snapshot.",
          "release_date": "2025-09-27",
          "status": "current",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-gene-enzprimers-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-gene-enzprimers-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official repository publishes the public questions and both public/private ID manifests.",
        "biosafety_notes": "BioBench Atlas stores metadata and aggregate results only and does not mirror question or sequence content.",
        "grader": "The official harness scores exact single-choice outputs and reports accuracy, precision, coverage, and n.",
        "level": "partially-open",
        "license": "CC-BY-SA-4.0",
        "tasks": "40 questions are public and 10 question contents are withheld."
      },
      "aliases": [
        "SeqQA — Gene-to-Gibson primers (HindIII)"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": null,
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "design",
        "scientific-reasoning"
      ],
      "coverage_notes": [],
      "domains": [
        "genomics",
        "molecular-cell-biology"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-seqqa-pcr-gene-gibshindprimers-creator-mcq"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-pcr-gene-gibshindprimers-evidence-paper",
          "locator": {
            "note": "Task definition, total count, creator protocol, and result row.",
            "type": "section",
            "value": "Table 1; Appendix Tables 2–6; task description for SeqQA_PCR-gene-gibshindprimers-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/kind",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-pcr-gene-gibshindprimers-evidence-repository",
          "locator": {
            "note": "Public/private/total 40/10/50.",
            "type": "repository-path",
            "value": "SeqQA/PCR-gene-gibshindprimers-v1-splits.json; SeqQA/PCR-gene-gibshindprimers-v1-public.jsonl where present; task.py at 998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "source_id": "lab-bench-seqqa-pcr-gene-gibshindprimers-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets"
          ]
        }
      ],
      "field_status": [],
      "id": "lab-bench-seqqa-pcr-gene-gibshindprimers",
      "implementations": [
        {
          "commit": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838",
          "framework": "labbench Python evaluator",
          "notes": "Official task loader and exact-choice evaluator integration.",
          "status": "official",
          "url": "https://github.com/Future-House/LAB-Bench/blob/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA/task.py"
        }
      ],
      "kind": "track",
      "latest_version": "repository-998a8e0",
      "modalities": [
        "text",
        "dna-rna-sequence"
      ],
      "name": "LAB-Bench SeqQA — Gene-to-Gibson primers (HindIII)",
      "organizations": [
        "FutureHouse"
      ],
      "parent_id": "lab-bench-seqqa",
      "release_date": "2024-07-14",
      "resources": [
        {
          "access_notes": "Commit-pinned official task files with prefix SeqQA/PCR-gene-gibshindprimers-v1.",
          "id": "lab-bench-seqqa-pcr-gene-gibshindprimers-repository-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-SA-4.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA",
            "value": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "type": "repository",
          "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "repository-998a8e0",
        "entries": [
          {
            "confidence": "high",
            "count": 50,
            "count_basis": "questions across public and private splits",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-seqqa-pcr-gene-gibshindprimers-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Gibson-assembly primer design.",
            "reporting_status": "reported",
            "task_type_id": "dna-sequence-design"
          }
        ],
        "notes": "Single-purpose formal LAB-Bench track.",
        "status": "complete"
      },
      "summary": "Selects primers for Gibson assembly into a HindIII-linearized vector.",
      "task_counts": {
        "basis": "questions across public and private splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "questions",
            "count": 40,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-pcr-gene-gibshindprimers-public",
            "label": "Public split",
            "notes": "Released in the official repository and Hugging Face dataset.",
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 10,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-pcr-gene-gibshindprimers-private",
            "label": "Private contamination-monitoring split",
            "notes": "IDs are published; question contents are withheld.",
            "reporting_status": "reported"
          }
        ],
        "total": 50
      },
      "task_formats": [
        "multiple choice"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task count and public/private membership independently recomputed from the pinned split manifest.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2024-07-17",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibshindprimers-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "lab-bench-seqqa-pcr-gene-gibshindprimers-paper-v3",
          "label": "paper-v3",
          "notes": "Creator-paper full task snapshot used for Tables 2–4.",
          "release_date": "2024-07-17",
          "status": "active",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-gene-gibshindprimers-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-gene-gibshindprimers-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibshindprimers-evidence-repository"
          ],
          "formal_tracks": [],
          "id": "lab-bench-seqqa-pcr-gene-gibshindprimers-repository-998a8e0",
          "label": "repository-998a8e0",
          "notes": "Current commit-pinned task snapshot.",
          "release_date": "2025-09-27",
          "status": "current",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-gene-gibshindprimers-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-gene-gibshindprimers-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official repository publishes the public questions and both public/private ID manifests.",
        "biosafety_notes": "BioBench Atlas stores metadata and aggregate results only and does not mirror question or sequence content.",
        "grader": "The official harness scores exact single-choice outputs and reports accuracy, precision, coverage, and n.",
        "level": "partially-open",
        "license": "CC-BY-SA-4.0",
        "tasks": "40 questions are public and 10 question contents are withheld."
      },
      "aliases": [
        "SeqQA — Gene-to-Gibson primers (SmaI)"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": null,
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "design",
        "scientific-reasoning"
      ],
      "coverage_notes": [],
      "domains": [
        "genomics",
        "molecular-cell-biology"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-seqqa-pcr-gene-gibssmaprimers-creator-mcq"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-pcr-gene-gibssmaprimers-evidence-paper",
          "locator": {
            "note": "Task definition, total count, creator protocol, and result row.",
            "type": "section",
            "value": "Table 1; Appendix Tables 2–6; task description for SeqQA_PCR-gene-gibssmaprimers-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/kind",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-pcr-gene-gibssmaprimers-evidence-repository",
          "locator": {
            "note": "Public/private/total 40/10/50.",
            "type": "repository-path",
            "value": "SeqQA/PCR-gene-gibssmaprimers-v1-splits.json; SeqQA/PCR-gene-gibssmaprimers-v1-public.jsonl where present; task.py at 998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "source_id": "lab-bench-seqqa-pcr-gene-gibssmaprimers-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets"
          ]
        }
      ],
      "field_status": [],
      "id": "lab-bench-seqqa-pcr-gene-gibssmaprimers",
      "implementations": [
        {
          "commit": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838",
          "framework": "labbench Python evaluator",
          "notes": "Official task loader and exact-choice evaluator integration.",
          "status": "official",
          "url": "https://github.com/Future-House/LAB-Bench/blob/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA/task.py"
        }
      ],
      "kind": "track",
      "latest_version": "repository-998a8e0",
      "modalities": [
        "text",
        "dna-rna-sequence"
      ],
      "name": "LAB-Bench SeqQA — Gene-to-Gibson primers (SmaI)",
      "organizations": [
        "FutureHouse"
      ],
      "parent_id": "lab-bench-seqqa",
      "release_date": "2024-07-14",
      "resources": [
        {
          "access_notes": "Commit-pinned official task files with prefix SeqQA/PCR-gene-gibssmaprimers-v1.",
          "id": "lab-bench-seqqa-pcr-gene-gibssmaprimers-repository-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-SA-4.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA",
            "value": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "type": "repository",
          "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "repository-998a8e0",
        "entries": [
          {
            "confidence": "high",
            "count": 50,
            "count_basis": "questions across public and private splits",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-seqqa-pcr-gene-gibssmaprimers-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Gibson-assembly primer design.",
            "reporting_status": "reported",
            "task_type_id": "dna-sequence-design"
          }
        ],
        "notes": "Single-purpose formal LAB-Bench track.",
        "status": "complete"
      },
      "summary": "Selects primers for Gibson assembly into a SmaI-linearized vector.",
      "task_counts": {
        "basis": "questions across public and private splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "questions",
            "count": 40,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-pcr-gene-gibssmaprimers-public",
            "label": "Public split",
            "notes": "Released in the official repository and Hugging Face dataset.",
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 10,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-pcr-gene-gibssmaprimers-private",
            "label": "Private contamination-monitoring split",
            "notes": "IDs are published; question contents are withheld.",
            "reporting_status": "reported"
          }
        ],
        "total": 50
      },
      "task_formats": [
        "multiple choice"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task count and public/private membership independently recomputed from the pinned split manifest.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2024-07-17",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibssmaprimers-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "lab-bench-seqqa-pcr-gene-gibssmaprimers-paper-v3",
          "label": "paper-v3",
          "notes": "Creator-paper full task snapshot used for Tables 2–4.",
          "release_date": "2024-07-17",
          "status": "active",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-gene-gibssmaprimers-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-gene-gibssmaprimers-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibssmaprimers-evidence-repository"
          ],
          "formal_tracks": [],
          "id": "lab-bench-seqqa-pcr-gene-gibssmaprimers-repository-998a8e0",
          "label": "repository-998a8e0",
          "notes": "Current commit-pinned task snapshot.",
          "release_date": "2025-09-27",
          "status": "current",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-gene-gibssmaprimers-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-gene-gibssmaprimers-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official repository publishes the public questions and both public/private ID manifests.",
        "biosafety_notes": "BioBench Atlas stores metadata and aggregate results only and does not mirror question or sequence content.",
        "grader": "The official harness scores exact single-choice outputs and reports accuracy, precision, coverage, and n.",
        "level": "partially-open",
        "license": "CC-BY-SA-4.0",
        "tasks": "40 questions are public and 10 question contents are withheld."
      },
      "aliases": [
        "SeqQA — Primers-to-restriction enzymes"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": null,
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "scientific-reasoning"
      ],
      "coverage_notes": [],
      "domains": [
        "genomics",
        "molecular-cell-biology"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-seqqa-pcr-geneprimers-enz-creator-mcq"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-pcr-geneprimers-enz-evidence-paper",
          "locator": {
            "note": "Task definition, total count, creator protocol, and result row.",
            "type": "section",
            "value": "Table 1; Appendix Tables 2–6; task description for SeqQA_PCR-geneprimers-enz-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/kind",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-pcr-geneprimers-enz-evidence-repository",
          "locator": {
            "note": "Public/private/total 40/10/50.",
            "type": "repository-path",
            "value": "SeqQA/PCR-geneprimers-enz-v1-splits.json; SeqQA/PCR-geneprimers-enz-v1-public.jsonl where present; task.py at 998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "source_id": "lab-bench-seqqa-pcr-geneprimers-enz-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets"
          ]
        }
      ],
      "field_status": [],
      "id": "lab-bench-seqqa-pcr-geneprimers-enz",
      "implementations": [
        {
          "commit": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838",
          "framework": "labbench Python evaluator",
          "notes": "Official task loader and exact-choice evaluator integration.",
          "status": "official",
          "url": "https://github.com/Future-House/LAB-Bench/blob/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA/task.py"
        }
      ],
      "kind": "track",
      "latest_version": "repository-998a8e0",
      "modalities": [
        "text",
        "dna-rna-sequence"
      ],
      "name": "LAB-Bench SeqQA — Primers-to-restriction enzymes",
      "organizations": [
        "FutureHouse"
      ],
      "parent_id": "lab-bench-seqqa",
      "release_date": "2024-07-14",
      "resources": [
        {
          "access_notes": "Commit-pinned official task files with prefix SeqQA/PCR-geneprimers-enz-v1.",
          "id": "lab-bench-seqqa-pcr-geneprimers-enz-repository-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-SA-4.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA",
            "value": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "type": "repository",
          "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "repository-998a8e0",
        "entries": [
          {
            "confidence": "high",
            "count": 50,
            "count_basis": "questions across public and private splits",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-seqqa-pcr-geneprimers-enz-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Infers restriction enzymes from a primer pair rather than generating a new sequence.",
            "reporting_status": "reported",
            "task_type_id": "dna-sequence-analysis"
          }
        ],
        "notes": "Single-purpose formal LAB-Bench sequence-analysis track.",
        "status": "complete"
      },
      "summary": "Infers restriction enzymes from a gene name and primer pair.",
      "task_counts": {
        "basis": "questions across public and private splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "questions",
            "count": 40,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-pcr-geneprimers-enz-public",
            "label": "Public split",
            "notes": "Released in the official repository and Hugging Face dataset.",
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 10,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-pcr-geneprimers-enz-private",
            "label": "Private contamination-monitoring split",
            "notes": "IDs are published; question contents are withheld.",
            "reporting_status": "reported"
          }
        ],
        "total": 50
      },
      "task_formats": [
        "multiple choice"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task count and public/private membership independently recomputed from the pinned split manifest.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2024-07-17",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-geneprimers-enz-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "lab-bench-seqqa-pcr-geneprimers-enz-paper-v3",
          "label": "paper-v3",
          "notes": "Creator-paper full task snapshot used for Tables 2–4.",
          "release_date": "2024-07-17",
          "status": "active",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-geneprimers-enz-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-geneprimers-enz-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-geneprimers-enz-evidence-repository"
          ],
          "formal_tracks": [],
          "id": "lab-bench-seqqa-pcr-geneprimers-enz-repository-998a8e0",
          "label": "repository-998a8e0",
          "notes": "Current commit-pinned task snapshot.",
          "release_date": "2025-09-27",
          "status": "current",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-geneprimers-enz-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-geneprimers-enz-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official repository publishes the public questions and both public/private ID manifests.",
        "biosafety_notes": "BioBench Atlas stores metadata and aggregate results only and does not mirror question or sequence content.",
        "grader": "The official harness scores exact single-choice outputs and reports accuracy, precision, coverage, and n.",
        "level": "partially-open",
        "license": "CC-BY-SA-4.0",
        "tasks": "40 questions are public and 10 question contents are withheld."
      },
      "aliases": [
        "SeqQA — Amplicon length to primers"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": null,
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "design",
        "data-analysis"
      ],
      "coverage_notes": [],
      "domains": [
        "genomics",
        "molecular-cell-biology"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-seqqa-pcr-len-primers-creator-mcq"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-pcr-len-primers-evidence-paper",
          "locator": {
            "note": "Task definition, total count, creator protocol, and result row.",
            "type": "section",
            "value": "Table 1; Appendix Tables 2–6; task description for SeqQA_PCR-len-primers-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/kind",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-pcr-len-primers-evidence-repository",
          "locator": {
            "note": "Public/private/total 40/10/50.",
            "type": "repository-path",
            "value": "SeqQA/PCR-len-primers-v1-splits.json; SeqQA/PCR-len-primers-v1-public.jsonl where present; task.py at 998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "source_id": "lab-bench-seqqa-pcr-len-primers-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets"
          ]
        }
      ],
      "field_status": [],
      "id": "lab-bench-seqqa-pcr-len-primers",
      "implementations": [
        {
          "commit": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838",
          "framework": "labbench Python evaluator",
          "notes": "Official task loader and exact-choice evaluator integration.",
          "status": "official",
          "url": "https://github.com/Future-House/LAB-Bench/blob/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA/task.py"
        }
      ],
      "kind": "track",
      "latest_version": "repository-998a8e0",
      "modalities": [
        "text",
        "dna-rna-sequence"
      ],
      "name": "LAB-Bench SeqQA — Amplicon length to primers",
      "organizations": [
        "FutureHouse"
      ],
      "parent_id": "lab-bench-seqqa",
      "release_date": "2024-07-14",
      "resources": [
        {
          "access_notes": "Commit-pinned official task files with prefix SeqQA/PCR-len-primers-v1.",
          "id": "lab-bench-seqqa-pcr-len-primers-repository-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-SA-4.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA",
            "value": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "type": "repository",
          "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "repository-998a8e0",
        "entries": [
          {
            "confidence": "high",
            "count": 50,
            "count_basis": "questions across public and private splits",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-seqqa-pcr-len-primers-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "PCR primer selection under an amplicon-length constraint.",
            "reporting_status": "reported",
            "task_type_id": "dna-sequence-design"
          }
        ],
        "notes": "Single-purpose formal LAB-Bench track.",
        "status": "complete"
      },
      "summary": "Selects primers that produce a requested amplicon length from a DNA template.",
      "task_counts": {
        "basis": "questions across public and private splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "questions",
            "count": 40,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-pcr-len-primers-public",
            "label": "Public split",
            "notes": "Released in the official repository and Hugging Face dataset.",
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 10,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-pcr-len-primers-private",
            "label": "Private contamination-monitoring split",
            "notes": "IDs are published; question contents are withheld.",
            "reporting_status": "reported"
          }
        ],
        "total": 50
      },
      "task_formats": [
        "multiple choice"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task count and public/private membership independently recomputed from the pinned split manifest.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2024-07-17",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-len-primers-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "lab-bench-seqqa-pcr-len-primers-paper-v3",
          "label": "paper-v3",
          "notes": "Creator-paper full task snapshot used for Tables 2–4.",
          "release_date": "2024-07-17",
          "status": "active",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-len-primers-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-len-primers-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-len-primers-evidence-repository"
          ],
          "formal_tracks": [],
          "id": "lab-bench-seqqa-pcr-len-primers-repository-998a8e0",
          "label": "repository-998a8e0",
          "notes": "Current commit-pinned task snapshot.",
          "release_date": "2025-09-27",
          "status": "current",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-len-primers-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-len-primers-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official repository publishes the public questions and both public/private ID manifests.",
        "biosafety_notes": "BioBench Atlas stores metadata and aggregate results only and does not mirror question or sequence content.",
        "grader": "The official harness scores exact single-choice outputs and reports accuracy, precision, coverage, and n.",
        "level": "partially-open",
        "license": "CC-BY-SA-4.0",
        "tasks": "40 questions are public and 10 question contents are withheld."
      },
      "aliases": [
        "SeqQA — Primers to amplicon length"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": null,
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "data-analysis",
        "scientific-reasoning"
      ],
      "coverage_notes": [],
      "domains": [
        "genomics",
        "molecular-cell-biology"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-seqqa-pcr-primers-len-creator-mcq"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-pcr-primers-len-evidence-paper",
          "locator": {
            "note": "Task definition, total count, creator protocol, and result row.",
            "type": "section",
            "value": "Table 1; Appendix Tables 2–6; task description for SeqQA_PCR-primers-len-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/kind",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-pcr-primers-len-evidence-repository",
          "locator": {
            "note": "Public/private/total 40/10/50.",
            "type": "repository-path",
            "value": "SeqQA/PCR-primers-len-v1-splits.json; SeqQA/PCR-primers-len-v1-public.jsonl where present; task.py at 998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "source_id": "lab-bench-seqqa-pcr-primers-len-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets"
          ]
        }
      ],
      "field_status": [],
      "id": "lab-bench-seqqa-pcr-primers-len",
      "implementations": [
        {
          "commit": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838",
          "framework": "labbench Python evaluator",
          "notes": "Official task loader and exact-choice evaluator integration.",
          "status": "official",
          "url": "https://github.com/Future-House/LAB-Bench/blob/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA/task.py"
        }
      ],
      "kind": "track",
      "latest_version": "repository-998a8e0",
      "modalities": [
        "text",
        "dna-rna-sequence"
      ],
      "name": "LAB-Bench SeqQA — Primers to amplicon length",
      "organizations": [
        "FutureHouse"
      ],
      "parent_id": "lab-bench-seqqa",
      "release_date": "2024-07-14",
      "resources": [
        {
          "access_notes": "Commit-pinned official task files with prefix SeqQA/PCR-primers-len-v1.",
          "id": "lab-bench-seqqa-pcr-primers-len-repository-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-SA-4.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA",
            "value": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "type": "repository",
          "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "repository-998a8e0",
        "entries": [
          {
            "confidence": "high",
            "count": 50,
            "count_basis": "questions across public and private splits",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-seqqa-pcr-primers-len-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Calculates an amplicon length from primers and a DNA template.",
            "reporting_status": "reported",
            "task_type_id": "dna-sequence-analysis"
          }
        ],
        "notes": "Single-purpose formal LAB-Bench sequence-analysis track.",
        "status": "complete"
      },
      "summary": "Calculates expected amplicon length from a primer pair and DNA template.",
      "task_counts": {
        "basis": "questions across public and private splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "questions",
            "count": 40,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-pcr-primers-len-public",
            "label": "Public split",
            "notes": "Released in the official repository and Hugging Face dataset.",
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 10,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-pcr-primers-len-private",
            "label": "Private contamination-monitoring split",
            "notes": "IDs are published; question contents are withheld.",
            "reporting_status": "reported"
          }
        ],
        "total": 50
      },
      "task_formats": [
        "multiple choice"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task count and public/private membership independently recomputed from the pinned split manifest.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2024-07-17",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-primers-len-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "lab-bench-seqqa-pcr-primers-len-paper-v3",
          "label": "paper-v3",
          "notes": "Creator-paper full task snapshot used for Tables 2–4.",
          "release_date": "2024-07-17",
          "status": "active",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-primers-len-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-primers-len-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-primers-len-evidence-repository"
          ],
          "formal_tracks": [],
          "id": "lab-bench-seqqa-pcr-primers-len-repository-998a8e0",
          "label": "repository-998a8e0",
          "notes": "Current commit-pinned task snapshot.",
          "release_date": "2025-09-27",
          "status": "current",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-primers-len-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-primers-len-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official repository publishes the public questions and both public/private ID manifests.",
        "biosafety_notes": "BioBench Atlas stores metadata and aggregate results only and does not mirror question or sequence content.",
        "grader": "The official harness scores exact single-choice outputs and reports accuracy, precision, coverage, and n.",
        "level": "partially-open",
        "license": "CC-BY-SA-4.0",
        "tasks": "40 questions are public and 10 question contents are withheld."
      },
      "aliases": [
        "SeqQA — Sequence-to-restriction primers"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": null,
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "design",
        "scientific-reasoning"
      ],
      "coverage_notes": [],
      "domains": [
        "genomics",
        "molecular-cell-biology"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-seqqa-pcr-seq-enzprimers-creator-mcq"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-pcr-seq-enzprimers-evidence-paper",
          "locator": {
            "note": "Task definition, total count, creator protocol, and result row.",
            "type": "section",
            "value": "Table 1; Appendix Tables 2–6; task description for SeqQA_PCR-seq-enzprimers-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/kind",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-pcr-seq-enzprimers-evidence-repository",
          "locator": {
            "note": "Public/private/total 40/10/50.",
            "type": "repository-path",
            "value": "SeqQA/PCR-seq-enzprimers-v1-splits.json; SeqQA/PCR-seq-enzprimers-v1-public.jsonl where present; task.py at 998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "source_id": "lab-bench-seqqa-pcr-seq-enzprimers-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets"
          ]
        }
      ],
      "field_status": [],
      "id": "lab-bench-seqqa-pcr-seq-enzprimers",
      "implementations": [
        {
          "commit": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838",
          "framework": "labbench Python evaluator",
          "notes": "Official task loader and exact-choice evaluator integration.",
          "status": "official",
          "url": "https://github.com/Future-House/LAB-Bench/blob/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA/task.py"
        }
      ],
      "kind": "track",
      "latest_version": "repository-998a8e0",
      "modalities": [
        "text",
        "dna-rna-sequence"
      ],
      "name": "LAB-Bench SeqQA — Sequence-to-restriction primers",
      "organizations": [
        "FutureHouse"
      ],
      "parent_id": "lab-bench-seqqa",
      "release_date": "2024-07-14",
      "resources": [
        {
          "access_notes": "Commit-pinned official task files with prefix SeqQA/PCR-seq-enzprimers-v1.",
          "id": "lab-bench-seqqa-pcr-seq-enzprimers-repository-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-SA-4.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA",
            "value": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "type": "repository",
          "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "repository-998a8e0",
        "entries": [
          {
            "confidence": "high",
            "count": 50,
            "count_basis": "questions across public and private splits",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-seqqa-pcr-seq-enzprimers-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Restriction-cloning primer design from a sequence.",
            "reporting_status": "reported",
            "task_type_id": "dna-sequence-design"
          }
        ],
        "notes": "Single-purpose formal LAB-Bench track.",
        "status": "complete"
      },
      "summary": "Selects restriction-cloning primers from an explicit gene sequence and enzyme pair.",
      "task_counts": {
        "basis": "questions across public and private splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "questions",
            "count": 40,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-pcr-seq-enzprimers-public",
            "label": "Public split",
            "notes": "Released in the official repository and Hugging Face dataset.",
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 10,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-pcr-seq-enzprimers-private",
            "label": "Private contamination-monitoring split",
            "notes": "IDs are published; question contents are withheld.",
            "reporting_status": "reported"
          }
        ],
        "total": 50
      },
      "task_formats": [
        "multiple choice"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task count and public/private membership independently recomputed from the pinned split manifest.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2024-07-17",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-enzprimers-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "lab-bench-seqqa-pcr-seq-enzprimers-paper-v3",
          "label": "paper-v3",
          "notes": "Creator-paper full task snapshot used for Tables 2–4.",
          "release_date": "2024-07-17",
          "status": "active",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-seq-enzprimers-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-seq-enzprimers-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-enzprimers-evidence-repository"
          ],
          "formal_tracks": [],
          "id": "lab-bench-seqqa-pcr-seq-enzprimers-repository-998a8e0",
          "label": "repository-998a8e0",
          "notes": "Current commit-pinned task snapshot.",
          "release_date": "2025-09-27",
          "status": "current",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-seq-enzprimers-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-seq-enzprimers-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official repository publishes the public questions and both public/private ID manifests.",
        "biosafety_notes": "BioBench Atlas stores metadata and aggregate results only and does not mirror question or sequence content.",
        "grader": "The official harness scores exact single-choice outputs and reports accuracy, precision, coverage, and n.",
        "level": "partially-open",
        "license": "CC-BY-SA-4.0",
        "tasks": "40 questions are public and 10 question contents are withheld."
      },
      "aliases": [
        "SeqQA — Amplicon sequence to primers"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": null,
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "design",
        "data-analysis"
      ],
      "coverage_notes": [],
      "domains": [
        "genomics",
        "molecular-cell-biology"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-seqqa-pcr-seq-primers-creator-mcq"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-pcr-seq-primers-evidence-paper",
          "locator": {
            "note": "Task definition, total count, creator protocol, and result row.",
            "type": "section",
            "value": "Table 1; Appendix Tables 2–6; task description for SeqQA_PCR-seq-primers-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/kind",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-pcr-seq-primers-evidence-repository",
          "locator": {
            "note": "Public/private/total 40/10/50.",
            "type": "repository-path",
            "value": "SeqQA/PCR-seq-primers-v1-splits.json; SeqQA/PCR-seq-primers-v1-public.jsonl where present; task.py at 998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "source_id": "lab-bench-seqqa-pcr-seq-primers-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets"
          ]
        }
      ],
      "field_status": [],
      "id": "lab-bench-seqqa-pcr-seq-primers",
      "implementations": [
        {
          "commit": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838",
          "framework": "labbench Python evaluator",
          "notes": "Official task loader and exact-choice evaluator integration.",
          "status": "official",
          "url": "https://github.com/Future-House/LAB-Bench/blob/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA/task.py"
        }
      ],
      "kind": "track",
      "latest_version": "repository-998a8e0",
      "modalities": [
        "text",
        "dna-rna-sequence"
      ],
      "name": "LAB-Bench SeqQA — Amplicon sequence to primers",
      "organizations": [
        "FutureHouse"
      ],
      "parent_id": "lab-bench-seqqa",
      "release_date": "2024-07-14",
      "resources": [
        {
          "access_notes": "Commit-pinned official task files with prefix SeqQA/PCR-seq-primers-v1.",
          "id": "lab-bench-seqqa-pcr-seq-primers-repository-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-SA-4.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA",
            "value": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "type": "repository",
          "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "repository-998a8e0",
        "entries": [
          {
            "confidence": "high",
            "count": 50,
            "count_basis": "questions across public and private splits",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-seqqa-pcr-seq-primers-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "PCR primer selection for a target amplicon.",
            "reporting_status": "reported",
            "task_type_id": "dna-sequence-design"
          }
        ],
        "notes": "Single-purpose formal LAB-Bench track.",
        "status": "complete"
      },
      "summary": "Selects primers that produce a requested amplicon sequence from a DNA template.",
      "task_counts": {
        "basis": "questions across public and private splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "questions",
            "count": 40,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-pcr-seq-primers-public",
            "label": "Public split",
            "notes": "Released in the official repository and Hugging Face dataset.",
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 10,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-pcr-seq-primers-private",
            "label": "Private contamination-monitoring split",
            "notes": "IDs are published; question contents are withheld.",
            "reporting_status": "reported"
          }
        ],
        "total": 50
      },
      "task_formats": [
        "multiple choice"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task count and public/private membership independently recomputed from the pinned split manifest.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2024-07-17",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-primers-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "lab-bench-seqqa-pcr-seq-primers-paper-v3",
          "label": "paper-v3",
          "notes": "Creator-paper full task snapshot used for Tables 2–4.",
          "release_date": "2024-07-17",
          "status": "active",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-seq-primers-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-seq-primers-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-primers-evidence-repository"
          ],
          "formal_tracks": [],
          "id": "lab-bench-seqqa-pcr-seq-primers-repository-998a8e0",
          "label": "repository-998a8e0",
          "notes": "Current commit-pinned task snapshot.",
          "release_date": "2025-09-27",
          "status": "current",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-seq-primers-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-pcr-seq-primers-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official repository publishes the public questions and both public/private ID manifests.",
        "biosafety_notes": "BioBench Atlas stores metadata and aggregate results only and does not mirror question or sequence content.",
        "grader": "The official harness scores exact single-choice outputs and reports accuracy, precision, coverage, and n.",
        "level": "partially-open",
        "license": "CC-BY-SA-4.0",
        "tasks": "40 questions are public and 10 question contents are withheld."
      },
      "aliases": [
        "GC-seq-percent"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": null,
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "data-analysis"
      ],
      "coverage_notes": [],
      "domains": [
        "genomics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-seqqa-prop-seq-gcpercent-creator-mcq"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-prop-seq-gcpercent-evidence-paper",
          "locator": {
            "note": "Task definition, total count, creator protocol, and result row.",
            "type": "section",
            "value": "Table 1; Appendix Tables 2–6; task description for SeqQA_Prop-seq-gcpercent-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/kind",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-prop-seq-gcpercent-evidence-repository",
          "locator": {
            "note": "Public/private/total 40/10/50.",
            "type": "repository-path",
            "value": "SeqQA/Prop-seq-gcpercent-v1-splits.json; SeqQA/Prop-seq-gcpercent-v1-public.jsonl where present; task.py at 998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "source_id": "lab-bench-seqqa-prop-seq-gcpercent-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets"
          ]
        }
      ],
      "field_status": [],
      "id": "lab-bench-seqqa-prop-seq-gcpercent",
      "implementations": [
        {
          "commit": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838",
          "framework": "labbench Python evaluator",
          "notes": "Official task loader and exact-choice evaluator integration.",
          "status": "official",
          "url": "https://github.com/Future-House/LAB-Bench/blob/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA/task.py"
        }
      ],
      "kind": "track",
      "latest_version": "repository-998a8e0",
      "modalities": [
        "text",
        "dna-rna-sequence"
      ],
      "name": "LAB-Bench SeqQA — GC percentage",
      "organizations": [
        "FutureHouse"
      ],
      "parent_id": "lab-bench-seqqa",
      "release_date": "2024-07-14",
      "resources": [
        {
          "access_notes": "Commit-pinned official task files with prefix SeqQA/Prop-seq-gcpercent-v1.",
          "id": "lab-bench-seqqa-prop-seq-gcpercent-repository-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-SA-4.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA",
            "value": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "type": "repository",
          "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "repository-998a8e0",
        "entries": [
          {
            "confidence": "high",
            "count": 50,
            "count_basis": "questions across public and private splits",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-seqqa-prop-seq-gcpercent-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Calculates a DNA sequence GC percentage.",
            "reporting_status": "reported",
            "task_type_id": "dna-sequence-analysis"
          }
        ],
        "notes": "Single-purpose formal LAB-Bench sequence-analysis track.",
        "status": "complete"
      },
      "summary": "Calculates the rounded GC percentage of a DNA sequence.",
      "task_counts": {
        "basis": "questions across public and private splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "questions",
            "count": 40,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-prop-seq-gcpercent-public",
            "label": "Public split",
            "notes": "Released in the official repository and Hugging Face dataset.",
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 10,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-prop-seq-gcpercent-private",
            "label": "Private contamination-monitoring split",
            "notes": "IDs are published; question contents are withheld.",
            "reporting_status": "reported"
          }
        ],
        "total": 50
      },
      "task_formats": [
        "multiple choice"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task count and public/private membership independently recomputed from the pinned split manifest.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2024-07-17",
          "evidence_ids": [
            "lab-bench-seqqa-prop-seq-gcpercent-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "lab-bench-seqqa-prop-seq-gcpercent-paper-v3",
          "label": "paper-v3",
          "notes": "Creator-paper full task snapshot used for Tables 2–4.",
          "release_date": "2024-07-17",
          "status": "active",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-prop-seq-gcpercent-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-prop-seq-gcpercent-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "lab-bench-seqqa-prop-seq-gcpercent-evidence-repository"
          ],
          "formal_tracks": [],
          "id": "lab-bench-seqqa-prop-seq-gcpercent-repository-998a8e0",
          "label": "repository-998a8e0",
          "notes": "Current commit-pinned task snapshot.",
          "release_date": "2025-09-27",
          "status": "current",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-prop-seq-gcpercent-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-prop-seq-gcpercent-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official repository publishes the public questions and both public/private ID manifests.",
        "biosafety_notes": "BioBench Atlas stores metadata and aggregate results only and does not mirror question or sequence content.",
        "grader": "The official harness scores exact single-choice outputs and reports accuracy, precision, coverage, and n.",
        "level": "partially-open",
        "license": "CC-BY-SA-4.0",
        "tasks": "40 questions are public and 10 question contents are withheld."
      },
      "aliases": [
        "SeqQA — Restriction-fragment lengths"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": null,
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "data-analysis",
        "scientific-reasoning"
      ],
      "coverage_notes": [],
      "domains": [
        "genomics",
        "molecular-cell-biology"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-seqqa-re-seq-lenfrags-creator-mcq"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-re-seq-lenfrags-evidence-paper",
          "locator": {
            "note": "Task definition, total count, creator protocol, and result row.",
            "type": "section",
            "value": "Table 1; Appendix Tables 2–6; task description for SeqQA_RE-seq-lenfrags-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/kind",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-re-seq-lenfrags-evidence-repository",
          "locator": {
            "note": "Public/private/total 40/10/50.",
            "type": "repository-path",
            "value": "SeqQA/RE-seq-lenfrags-v1-splits.json; SeqQA/RE-seq-lenfrags-v1-public.jsonl where present; task.py at 998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "source_id": "lab-bench-seqqa-re-seq-lenfrags-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets"
          ]
        }
      ],
      "field_status": [],
      "id": "lab-bench-seqqa-re-seq-lenfrags",
      "implementations": [
        {
          "commit": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838",
          "framework": "labbench Python evaluator",
          "notes": "Official task loader and exact-choice evaluator integration.",
          "status": "official",
          "url": "https://github.com/Future-House/LAB-Bench/blob/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA/task.py"
        }
      ],
      "kind": "track",
      "latest_version": "repository-998a8e0",
      "modalities": [
        "text",
        "dna-rna-sequence"
      ],
      "name": "LAB-Bench SeqQA — Restriction-fragment lengths",
      "organizations": [
        "FutureHouse"
      ],
      "parent_id": "lab-bench-seqqa",
      "release_date": "2024-07-14",
      "resources": [
        {
          "access_notes": "Commit-pinned official task files with prefix SeqQA/RE-seq-lenfrags-v1.",
          "id": "lab-bench-seqqa-re-seq-lenfrags-repository-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-SA-4.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA",
            "value": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "type": "repository",
          "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "repository-998a8e0",
        "entries": [
          {
            "confidence": "high",
            "count": 50,
            "count_basis": "questions across public and private splits",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-seqqa-re-seq-lenfrags-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Calculates restriction-fragment lengths.",
            "reporting_status": "reported",
            "task_type_id": "dna-sequence-analysis"
          }
        ],
        "notes": "Single-purpose formal LAB-Bench sequence-analysis track.",
        "status": "complete"
      },
      "summary": "Calculates fragment lengths after restriction digestion of a DNA sequence.",
      "task_counts": {
        "basis": "questions across public and private splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "questions",
            "count": 40,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-re-seq-lenfrags-public",
            "label": "Public split",
            "notes": "Released in the official repository and Hugging Face dataset.",
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 10,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-re-seq-lenfrags-private",
            "label": "Private contamination-monitoring split",
            "notes": "IDs are published; question contents are withheld.",
            "reporting_status": "reported"
          }
        ],
        "total": 50
      },
      "task_formats": [
        "multiple choice"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task count and public/private membership independently recomputed from the pinned split manifest.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2024-07-17",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-lenfrags-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "lab-bench-seqqa-re-seq-lenfrags-paper-v3",
          "label": "paper-v3",
          "notes": "Creator-paper full task snapshot used for Tables 2–4.",
          "release_date": "2024-07-17",
          "status": "active",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-re-seq-lenfrags-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-re-seq-lenfrags-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-lenfrags-evidence-repository"
          ],
          "formal_tracks": [],
          "id": "lab-bench-seqqa-re-seq-lenfrags-repository-998a8e0",
          "label": "repository-998a8e0",
          "notes": "Current commit-pinned task snapshot.",
          "release_date": "2025-09-27",
          "status": "current",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-re-seq-lenfrags-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-re-seq-lenfrags-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official repository publishes the public questions and both public/private ID manifests.",
        "biosafety_notes": "BioBench Atlas stores metadata and aggregate results only and does not mirror question or sequence content.",
        "grader": "The official harness scores exact single-choice outputs and reports accuracy, precision, coverage, and n.",
        "level": "partially-open",
        "license": "CC-BY-SA-4.0",
        "tasks": "40 questions are public and 10 question contents are withheld."
      },
      "aliases": [
        "SeqQA — Restriction-fragment count"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": null,
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "data-analysis",
        "scientific-reasoning"
      ],
      "coverage_notes": [],
      "domains": [
        "genomics",
        "molecular-cell-biology"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-seqqa-re-seq-numfrags-creator-mcq"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-re-seq-numfrags-evidence-paper",
          "locator": {
            "note": "Task definition, total count, creator protocol, and result row.",
            "type": "section",
            "value": "Table 1; Appendix Tables 2–6; task description for SeqQA_RE-seq-numfrags-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/kind",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-re-seq-numfrags-evidence-repository",
          "locator": {
            "note": "Public/private/total 40/10/50.",
            "type": "repository-path",
            "value": "SeqQA/RE-seq-numfrags-v1-splits.json; SeqQA/RE-seq-numfrags-v1-public.jsonl where present; task.py at 998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "source_id": "lab-bench-seqqa-re-seq-numfrags-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets"
          ]
        }
      ],
      "field_status": [],
      "id": "lab-bench-seqqa-re-seq-numfrags",
      "implementations": [
        {
          "commit": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838",
          "framework": "labbench Python evaluator",
          "notes": "Official task loader and exact-choice evaluator integration.",
          "status": "official",
          "url": "https://github.com/Future-House/LAB-Bench/blob/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA/task.py"
        }
      ],
      "kind": "track",
      "latest_version": "repository-998a8e0",
      "modalities": [
        "text",
        "dna-rna-sequence"
      ],
      "name": "LAB-Bench SeqQA — Restriction-fragment count",
      "organizations": [
        "FutureHouse"
      ],
      "parent_id": "lab-bench-seqqa",
      "release_date": "2024-07-14",
      "resources": [
        {
          "access_notes": "Commit-pinned official task files with prefix SeqQA/RE-seq-numfrags-v1.",
          "id": "lab-bench-seqqa-re-seq-numfrags-repository-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-SA-4.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA",
            "value": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "type": "repository",
          "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SeqQA"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "repository-998a8e0",
        "entries": [
          {
            "confidence": "high",
            "count": 50,
            "count_basis": "questions across public and private splits",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-seqqa-re-seq-numfrags-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Calculates a restriction-fragment count.",
            "reporting_status": "reported",
            "task_type_id": "dna-sequence-analysis"
          }
        ],
        "notes": "Single-purpose formal LAB-Bench sequence-analysis track.",
        "status": "complete"
      },
      "summary": "Calculates the number of fragments after restriction digestion of a DNA sequence.",
      "task_counts": {
        "basis": "questions across public and private splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "questions",
            "count": 40,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-re-seq-numfrags-public",
            "label": "Public split",
            "notes": "Released in the official repository and Hugging Face dataset.",
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 10,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-seqqa-re-seq-numfrags-private",
            "label": "Private contamination-monitoring split",
            "notes": "IDs are published; question contents are withheld.",
            "reporting_status": "reported"
          }
        ],
        "total": 50
      },
      "task_formats": [
        "multiple choice"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task count and public/private membership independently recomputed from the pinned split manifest.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2024-07-17",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-numfrags-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "lab-bench-seqqa-re-seq-numfrags-paper-v3",
          "label": "paper-v3",
          "notes": "Creator-paper full task snapshot used for Tables 2–4.",
          "release_date": "2024-07-17",
          "status": "active",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-re-seq-numfrags-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-re-seq-numfrags-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-numfrags-evidence-repository"
          ],
          "formal_tracks": [],
          "id": "lab-bench-seqqa-re-seq-numfrags-repository-998a8e0",
          "label": "repository-998a8e0",
          "notes": "Current commit-pinned task snapshot.",
          "release_date": "2025-09-27",
          "status": "current",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-re-seq-numfrags-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 10,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-seqqa-re-seq-numfrags-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 50
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official repository publishes the public questions and both public/private ID manifests.",
        "biosafety_notes": "BioBench Atlas stores metadata and aggregate results only and does not mirror question or sequence content.",
        "grader": "The official harness scores exact single-choice outputs and reports accuracy, precision, coverage, and n.",
        "level": "partially-open",
        "license": "CC-BY-SA-4.0",
        "tasks": "82 questions are public and 20 question contents are withheld."
      },
      "aliases": [
        "SuppQA"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": null,
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "retrieval",
        "evidence-synthesis",
        "scientific-reasoning"
      ],
      "coverage_notes": [],
      "domains": [
        "life-science"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-suppqa-creator-mcq"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-suppqa-evidence-paper",
          "locator": {
            "note": "Task definition, total count, creator protocol, and result row.",
            "type": "section",
            "value": "Table 1; Appendix Tables 2–6; task description for SuppQA"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/kind",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-suppqa-evidence-repository",
          "locator": {
            "note": "Public/private/total 82/20/102.",
            "type": "repository-path",
            "value": "SuppQA/suppqa-v1-splits.json; SuppQA/suppqa-v1-public.jsonl where present; task.py at 998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "source_id": "lab-bench-suppqa-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets"
          ]
        }
      ],
      "field_status": [],
      "id": "lab-bench-suppqa",
      "implementations": [
        {
          "commit": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838",
          "framework": "labbench Python evaluator",
          "notes": "Official task loader and exact-choice evaluator integration.",
          "status": "official",
          "url": "https://github.com/Future-House/LAB-Bench/blob/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SuppQA/task.py"
        }
      ],
      "kind": "track",
      "latest_version": "repository-998a8e0",
      "modalities": [
        "text",
        "paper",
        "web"
      ],
      "name": "LAB-Bench SuppQA",
      "organizations": [
        "FutureHouse"
      ],
      "parent_id": "lab-bench",
      "release_date": "2024-07-14",
      "resources": [
        {
          "access_notes": "Commit-pinned official task files with prefix SuppQA/suppqa-v1.",
          "id": "lab-bench-suppqa-repository-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-SA-4.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SuppQA",
            "value": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "type": "repository",
          "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/SuppQA"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "repository-998a8e0",
        "entries": [
          {
            "confidence": "high",
            "count": 102,
            "count_basis": "questions across public and private splits",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-suppqa-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Supplementary text and table interpretation.",
            "reporting_status": "reported",
            "task_type_id": "scientific-evidence-interpretation"
          }
        ],
        "notes": "Single-purpose formal LAB-Bench track.",
        "status": "complete"
      },
      "summary": "Retrieval and interpretation questions answerable from paper supplementary text or PDF tables.",
      "task_counts": {
        "basis": "questions across public and private splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "questions",
            "count": 82,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-suppqa-public",
            "label": "Public split",
            "notes": "Released in the official repository and Hugging Face dataset.",
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 20,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-suppqa-private",
            "label": "Private contamination-monitoring split",
            "notes": "IDs are published; question contents are withheld.",
            "reporting_status": "reported"
          }
        ],
        "total": 102
      },
      "task_formats": [
        "multiple choice"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task count and public/private membership independently recomputed from the pinned split manifest.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2024-07-17",
          "evidence_ids": [
            "lab-bench-suppqa-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "lab-bench-suppqa-paper-v3",
          "label": "paper-v3",
          "notes": "Creator-paper full task snapshot used for Tables 2–4.",
          "release_date": "2024-07-17",
          "status": "active",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 82,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-suppqa-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 20,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-suppqa-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 102
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "lab-bench-suppqa-evidence-repository"
          ],
          "formal_tracks": [],
          "id": "lab-bench-suppqa-repository-998a8e0",
          "label": "repository-998a8e0",
          "notes": "Current commit-pinned task snapshot.",
          "release_date": "2025-09-27",
          "status": "current",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 82,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-suppqa-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 20,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-suppqa-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 102
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The official repository publishes the public questions and both public/private ID manifests.",
        "biosafety_notes": "BioBench Atlas stores metadata and aggregate results only and does not mirror question or sequence content.",
        "grader": "The official harness scores exact single-choice outputs and reports accuracy, precision, coverage, and n.",
        "level": "partially-open",
        "license": "CC-BY-SA-4.0",
        "tasks": "244 questions are public and 61 question contents are withheld."
      },
      "aliases": [
        "TableQA"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": null,
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "evidence-synthesis",
        "data-analysis",
        "scientific-reasoning"
      ],
      "coverage_notes": [],
      "domains": [
        "life-science"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-tableqa-creator-mcq"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-tableqa-evidence-paper",
          "locator": {
            "note": "Task definition, total count, creator protocol, and result row.",
            "type": "section",
            "value": "Table 1; Appendix Tables 2–6; task description for TableQA"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/kind",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-tableqa-evidence-repository",
          "locator": {
            "note": "Public/private/total 244/61/305.",
            "type": "repository-path",
            "value": "TableQA/tableqa-v1-splits.json; TableQA/tableqa-v1-public.jsonl where present; task.py at 998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "source_id": "lab-bench-tableqa-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets"
          ]
        }
      ],
      "field_status": [],
      "id": "lab-bench-tableqa",
      "implementations": [
        {
          "commit": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838",
          "framework": "labbench Python evaluator",
          "notes": "Official task loader and exact-choice evaluator integration.",
          "status": "official",
          "url": "https://github.com/Future-House/LAB-Bench/blob/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/TableQA/task.py"
        }
      ],
      "kind": "track",
      "latest_version": "repository-998a8e0",
      "modalities": [
        "text",
        "table",
        "image"
      ],
      "name": "LAB-Bench TableQA",
      "organizations": [
        "FutureHouse"
      ],
      "parent_id": "lab-bench",
      "release_date": "2024-07-14",
      "resources": [
        {
          "access_notes": "Commit-pinned official task files with prefix TableQA/tableqa-v1.",
          "id": "lab-bench-tableqa-repository-resource",
          "last_checked": "2026-07-21",
          "license": "CC-BY-SA-4.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/TableQA",
            "value": "998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838"
          },
          "type": "repository",
          "url": "https://github.com/Future-House/LAB-Bench/tree/998a8e0a40cf116c80e1b0e7a805ebb5fb9fa838/TableQA"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "repository-998a8e0",
        "entries": [
          {
            "confidence": "high",
            "count": 305,
            "count_basis": "questions across public and private splits",
            "count_ref": "/task_counts/total",
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lab-bench-tableqa-evidence-paper"
            ],
            "mapping_method": "official-track",
            "notes": "Scientific table interpretation.",
            "reporting_status": "reported",
            "task_type_id": "scientific-evidence-interpretation"
          }
        ],
        "notes": "Single-purpose formal LAB-Bench track.",
        "status": "complete"
      },
      "summary": "Lookup, calculation, and reasoning questions over table images extracted from scientific papers.",
      "task_counts": {
        "basis": "questions across public and private splits",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "questions",
            "count": 244,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-tableqa-public",
            "label": "Public split",
            "notes": "Released in the official repository and Hugging Face dataset.",
            "reporting_status": "reported"
          },
          {
            "basis": "questions",
            "count": 61,
            "exclusive": true,
            "exhaustive": true,
            "id": "lab-bench-tableqa-private",
            "label": "Private contamination-monitoring split",
            "notes": "IDs are published; question contents are withheld.",
            "reporting_status": "reported"
          }
        ],
        "total": 305
      },
      "task_formats": [
        "multiple choice"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Task count and public/private membership independently recomputed from the pinned split manifest.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2024-07-17",
          "evidence_ids": [
            "lab-bench-tableqa-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "lab-bench-tableqa-paper-v3",
          "label": "paper-v3",
          "notes": "Creator-paper full task snapshot used for Tables 2–4.",
          "release_date": "2024-07-17",
          "status": "active",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 244,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-tableqa-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 61,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-tableqa-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 305
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "lab-bench-tableqa-evidence-repository"
          ],
          "formal_tracks": [],
          "id": "lab-bench-tableqa-repository-998a8e0",
          "label": "repository-998a8e0",
          "notes": "Current commit-pinned task snapshot.",
          "release_date": "2025-09-27",
          "status": "current",
          "task_counts": {
            "basis": "questions across public and private splits",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 244,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-tableqa-public",
                "label": "Public split",
                "notes": "Released in the official repository and Hugging Face dataset.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions",
                "count": 61,
                "exclusive": true,
                "exhaustive": true,
                "id": "lab-bench-tableqa-private",
                "label": "Private contamination-monitoring split",
                "notes": "IDs are published; question contents are withheld.",
                "reporting_status": "reported"
              }
            ],
            "total": 305
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "Selected examples are public, but the complete set of 1,062 artifacts is restricted.",
        "biosafety_notes": "The report says public materials may be restricted for licensing, privacy, proprietary-information, or biological-safety reasons; this registry reproduces metadata only.",
        "grader": "Full task-specific expert rubrics are not released; selected example rubrics are public.",
        "level": "private-internal",
        "license": null,
        "tasks": "The complete task set is not publicly downloadable; selected examples are public."
      },
      "aliases": [
        "Life Science Bench"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Official sources do not provide a numbered upstream version or standalone binding-type counts; these are represented as an initial-release snapshot and explicit Not reported values, not inferred numbers.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "evidence-synthesis",
        "retrieval",
        "design",
        "generation",
        "optimization",
        "data-analysis",
        "experiment-planning",
        "troubleshooting",
        "scientific-reasoning",
        "scientific-communication"
      ],
      "coverage_notes": [
        {
          "count": 136,
          "coverage": "observed",
          "notes": "Figure 13 column total for the primary Protein + Structural Biology domain.",
          "reporting_status": "reported",
          "tag": "protein-science"
        },
        {
          "count": 62,
          "coverage": "observed",
          "notes": "Figure 13 cross-tabulation of the protein domain and design/optimization workflow.",
          "reporting_status": "reported",
          "tag": "protein-design"
        },
        {
          "count": null,
          "coverage": "explicitly-in-scope",
          "notes": "The official protein-domain definition includes binding and a protein-protein binding example, but no standalone binding-type count is published.",
          "reporting_status": "not_reported",
          "tag": "protein-protein-binding"
        },
        {
          "count": null,
          "coverage": "explicitly-in-scope",
          "notes": "The official protein-domain definition includes binding and ligand-related examples, but no standalone binding-type count is published.",
          "reporting_status": "not_reported",
          "tag": "protein-ligand-binding"
        }
      ],
      "domains": [
        "life-science",
        "protein-science",
        "protein-sequence",
        "protein-structure",
        "protein-design",
        "protein-protein-binding",
        "protein-ligand-binding",
        "genomics",
        "transcriptomics",
        "spatial-omics",
        "molecular-cell-biology",
        "assay-screening",
        "bioinformatics",
        "medchem",
        "clinical-translational"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "lifescibench-preprint"
      ],
      "evaluation_run_ids": [
        "lifescibench-official-full"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lifescibench-evidence-release",
          "locator": {
            "note": "Official release page.",
            "type": "section",
            "value": "Release date; introduction; What LifeSciBench measures; Dataset construction; Grading and rubric breakdown"
          },
          "source_id": "lifescibench-launch-resource",
          "source_type": "resource",
          "supports": [
            "/release_date",
            "/latest_version",
            "/resources"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lifescibench-evidence-identity",
          "locator": {
            "note": "Official identity, scope, task structure, and creator affiliations.",
            "type": "section",
            "value": "pp. 1–4, title, affiliations, abstract, introduction, and Sections 3.1–3.3"
          },
          "source_id": "lifescibench-preprint",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/kind",
            "/summary",
            "/task_formats"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lifescibench-evidence-counts",
          "locator": {
            "note": "Explicit labels report 750 total, 136 Protein + Structural Biology, and 62 at Design and optimization × Protein + Structural Biology.",
            "type": "figure",
            "value": "pp. 6 and 18, Table 2 and Figure 13"
          },
          "source_id": "lifescibench-preprint",
          "source_type": "work",
          "supports": [
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes/0",
            "/coverage_notes/1",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lifescibench-evidence-taxonomy",
          "locator": {
            "note": "Official workflow, domain, evidence-source taxonomies and public examples.",
            "type": "section",
            "value": "pp. 3–4 and 17–18, Sections 3.1–3.3 and Appendix B.1–B.4; pp. 20–24, example tasks"
          },
          "source_id": "lifescibench-preprint",
          "source_type": "work",
          "supports": [
            "/domains",
            "/capabilities",
            "/modalities",
            "/coverage_notes/2",
            "/coverage_notes/3",
            "/scientific_task_classification/entries/1",
            "/scientific_task_classification/entries/2"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lifescibench-evidence-access",
          "locator": {
            "note": "Describes licensing, privacy, proprietary-information, and biosafety release restrictions.",
            "type": "section",
            "value": "p. 16, Appendix A.5 Data Availability and Safety Disclosure"
          },
          "source_id": "lifescibench-preprint",
          "source_type": "work",
          "supports": [
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes"
          ]
        }
      ],
      "field_status": [],
      "id": "lifescibench",
      "implementations": [
        {
          "commit": null,
          "framework": "official internal harness",
          "notes": "No public runnable harness was identified in the official paper or release page.",
          "status": "not-available",
          "url": null
        }
      ],
      "kind": "agentic-eval",
      "latest_version": "initial-release",
      "modalities": [
        "text",
        "paper",
        "table",
        "figure",
        "image",
        "dna-rna-sequence",
        "protein-sequence",
        "structure-3d",
        "raw-omics",
        "web",
        "wet-lab-output"
      ],
      "name": "LifeSciBench",
      "organizations": [
        "OpenAI",
        "Tacit Labs"
      ],
      "parent_id": null,
      "release_date": "2026-06-17",
      "resources": [
        {
          "access_notes": "Public official preprint; task and artifact corpus is not included.",
          "id": "lifescibench-paper-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "snapshot",
            "url": "https://cdn.openai.com/pdf/b4299379-0a97-4ffa-8b9b-c3fbb299caa9/lifescibench_preprint.pdf",
            "value": "sha256:830d366ed65061a7c2b404e7f93231cf96f44ff5fef702c268d7296ed8edfb9b"
          },
          "type": "paper",
          "url": "https://cdn.openai.com/pdf/b4299379-0a97-4ffa-8b9b-c3fbb299caa9/lifescibench_preprint.pdf"
        },
        {
          "access_notes": "Official release page with selected tasks, summary counts, grading description, and results.",
          "id": "lifescibench-launch-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": null,
          "type": "website",
          "url": "https://openai.com/index/introducing-life-sci-bench/"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "initial-release",
        "entries": [
          {
            "confidence": "high",
            "count": 62,
            "count_basis": "Expert-authored tasks in the Protein primary domain and Design / Optimization workflow cell.",
            "count_ref": "/coverage_notes/1/count",
            "count_unit": "tasks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lifescibench-evidence-counts"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "This is a broad design-or-optimization count; it is not relabeled as sequence generation.",
            "reporting_status": "reported",
            "task_type_id": "protein-design"
          },
          {
            "confidence": "high",
            "count": null,
            "count_basis": "Expert-authored benchmark tasks.",
            "count_ref": null,
            "count_unit": "tasks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lifescibench-evidence-taxonomy"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "The official protein-domain definition and examples include protein-protein binding, without a standalone count.",
            "reporting_status": "not_reported",
            "task_type_id": "protein-protein-interaction-prediction"
          },
          {
            "confidence": "high",
            "count": null,
            "count_basis": "Expert-authored benchmark tasks.",
            "count_ref": null,
            "count_unit": "tasks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "lifescibench-evidence-taxonomy"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "The source does not distinguish pose from affinity, so the broad task is retained.",
            "reporting_status": "not_reported",
            "task_type_id": "protein-ligand-binding-prediction"
          }
        ],
        "notes": "Official sources identify broad protein design and binding coverage but do not publish an exhaustive task-level taxonomy or binding subtype counts.",
        "status": "partial"
      },
      "summary": "Expert-authored, artifact-rich free-response tasks that evaluate realistic research judgment across applied life-science workflows.",
      "task_counts": {
        "basis": "expert-authored tasks",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "tasks",
            "count": 136,
            "exclusive": false,
            "exhaustive": false,
            "id": "protein-primary-domain",
            "label": "Protein and structural biology primary domain",
            "notes": "Figure 13 column total for the Protein + Structural Biology domain.",
            "reporting_status": "reported"
          },
          {
            "basis": "tasks",
            "count": 62,
            "exclusive": false,
            "exhaustive": false,
            "id": "protein-design-optimization",
            "label": "Protein-domain tasks in design and optimization workflow",
            "notes": "Figure 13 cell at Design and optimization × Protein + Structural Biology.",
            "reporting_status": "reported"
          }
        ],
        "total": 750
      },
      "task_formats": [
        "expert free response",
        "artifact-grounded analysis"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Field-level audit completed against the official OpenAI preprint and release page.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-06-17",
          "evidence_ids": [
            "lifescibench-evidence-release",
            "lifescibench-evidence-counts"
          ],
          "formal_tracks": [],
          "id": "lifescibench-initial-release",
          "label": "initial-release",
          "notes": "Registry-defined label for the June 17, 2026 initial release; the official sources do not report a numbered upstream version.",
          "release_date": "2026-06-17",
          "status": "current",
          "task_counts": {
            "basis": "expert-authored tasks",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "tasks",
                "count": 136,
                "exclusive": false,
                "exhaustive": false,
                "id": "protein-primary-domain",
                "label": "Protein and structural biology primary domain",
                "notes": "Figure 13 column total for the Protein + Structural Biology domain.",
                "reporting_status": "reported"
              },
              {
                "basis": "tasks",
                "count": 62,
                "exclusive": false,
                "exhaustive": false,
                "id": "protein-design-optimization",
                "label": "Protein-domain tasks in design and optimization workflow",
                "notes": "Figure 13 cell at Design and optimization × Protein + Structural Biology.",
                "reporting_status": "reported"
              }
            ],
            "total": 750
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "DeepChem publishes featurizers, splitters, model implementations, and benchmark utilities.",
        "biosafety_notes": "The suite includes drug activity and toxicity endpoints; this registry mirrors no compounds or labels.",
        "grader": "Deterministic regression and classification metrics with dataset-specific recommended splits.",
        "level": "fully-open",
        "license": "MIT for DeepChem code; the Chemical Science paper is CC BY-NC 3.0 and underlying datasets retain source-specific terms.",
        "tasks": "The original public datasets and standard loaders are distributed through DeepChem."
      },
      "aliases": [
        "MoleculeNet original suite"
      ],
      "audit": {
        "audited_date": "2026-07-22",
        "notes": "Original 17-collection scope is separated from the larger living DeepChem catalog; endpoint counts are not conflated with dataset counts.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [
        "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-1-use",
        "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-2-use",
        "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-3-use",
        "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-4-use"
      ],
      "capabilities": [
        "prediction",
        "classification",
        "regression"
      ],
      "coverage_notes": [
        {
          "count": 17,
          "coverage": "explicitly-in-scope",
          "notes": "Dataset collections cover molecular properties and drug-discovery endpoints; 17 is not the number of individual prediction endpoints.",
          "reporting_status": "reported",
          "tag": "medchem"
        },
        {
          "count": 1,
          "coverage": "explicitly-in-scope",
          "notes": "PDBbind is the explicit binding-affinity collection; other biophysical activity collections are not relabeled as affinity tasks.",
          "reporting_status": "reported",
          "tag": "protein-ligand-binding"
        }
      ],
      "domains": [
        "medchem",
        "protein-ligand-binding",
        "assay-screening",
        "clinical-translational"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "enhancing-molecular-property-prediction-with-auxiliary",
        "moleculenet-paper"
      ],
      "evaluation_run_ids": [
        "moleculenet-creator-full"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "moleculenet-paper-definition-evidence",
          "locator": {
            "note": "Defines 17 collections, four categories, over 800 endpoints, 80/10/10 splits, recommended splitters, metrics, and creator baselines.",
            "type": "table",
            "value": "Sections 3-5, Figure 2 and Tables 1-3"
          },
          "source_id": "moleculenet-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/resources",
            "/versions/0/release_date",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0",
            "/scientific_task_classification/entries/1",
            "/scientific_task_classification/entries/2"
          ]
        },
        {
          "accessed_date": "2026-07-22",
          "id": "moleculenet-repository-evidence",
          "locator": {
            "note": "Pins the maintained loader and evaluation implementation while preserving the original-paper version boundary.",
            "type": "repository-path",
            "value": "deepchem/molnet; LICENSE at fbe3b911a94a6eb8c84c1a9e7ac67a472c49ea80"
          },
          "source_id": "moleculenet-deepchem-resource",
          "source_type": "resource",
          "supports": [
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/implementations",
            "/versions/0/formal_tracks"
          ]
        }
      ],
      "field_status": [],
      "id": "moleculenet",
      "implementations": [
        {
          "commit": "fbe3b911a94a6eb8c84c1a9e7ac67a472c49ea80",
          "framework": "DeepChem MolNet",
          "notes": "Current maintained loaders; users must select the original datasets and paper-recommended split/metric to reproduce the original suite.",
          "status": "official",
          "url": "https://github.com/deepchem/deepchem/tree/fbe3b911a94a6eb8c84c1a9e7ac67a472c49ea80/deepchem/molnet"
        }
      ],
      "kind": "suite",
      "latest_version": "original-2017",
      "modalities": [
        "small-molecule-structure",
        "structure-3d"
      ],
      "name": "MoleculeNet",
      "organizations": [
        "Stanford University",
        "DeepChem"
      ],
      "parent_id": null,
      "release_date": "2017-03-02",
      "resources": [
        {
          "access_notes": "Peer-reviewed creator paper in Chemical Science.",
          "id": "moleculenet-paper-resource",
          "last_checked": "2026-07-22",
          "license": "CC BY-NC 3.0",
          "pin": null,
          "type": "paper",
          "url": "https://doi.org/10.1039/C7SC02664A"
        },
        {
          "access_notes": "Official maintained implementation; the record intentionally retains the original 17-collection paper snapshot rather than all current MolNet loaders.",
          "id": "moleculenet-deepchem-resource",
          "last_checked": "2026-07-22",
          "license": "MIT",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/deepchem/deepchem/tree/fbe3b911a94a6eb8c84c1a9e7ac67a472c49ea80",
            "value": "fbe3b911a94a6eb8c84c1a9e7ac67a472c49ea80"
          },
          "type": "repository",
          "url": "https://github.com/deepchem/deepchem"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "original-2017",
        "entries": [
          {
            "confidence": "high",
            "count": 17,
            "count_basis": "Original paper dataset collections.",
            "count_ref": "/task_counts/total",
            "count_unit": "other",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "moleculenet-paper-definition-evidence"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "Umbrella mapping for all 17 original dataset collections, which span quantum, physical, biophysical, and physiological properties.",
            "reporting_status": "reported",
            "task_type_id": "small-molecule-property-prediction"
          },
          {
            "confidence": "high",
            "count": 5,
            "count_basis": "Original paper dataset collections.",
            "count_ref": null,
            "count_unit": "other",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "moleculenet-paper-definition-evidence"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "The five physiology collections are ESOL-independent ADMET or toxicity datasets; no endpoint-level total is asserted here.",
            "reporting_status": "reported",
            "task_type_id": "admet-toxicity-prediction"
          },
          {
            "confidence": "high",
            "count": 1,
            "count_basis": "Original paper dataset collections.",
            "count_ref": null,
            "count_unit": "other",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "moleculenet-paper-definition-evidence"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "PDBbind is the explicitly structure-based protein-ligand affinity collection.",
            "reporting_status": "reported",
            "task_type_id": "protein-ligand-binding-affinity"
          }
        ],
        "notes": "The fixed original suite is mapped at dataset-collection level; heterogeneous biochemical endpoint collections are not atomized into unsupported task counts.",
        "status": "partial"
      },
      "summary": "The original molecular-machine-learning benchmark of 17 dataset collections and more than 800 prediction endpoints spanning quantum, physicochemical, biophysical, and physiological properties.",
      "task_counts": {
        "basis": "original paper dataset collections",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "original paper dataset collections",
            "count": 4,
            "exclusive": true,
            "exhaustive": true,
            "id": "moleculenet-quantum",
            "label": "Quantum mechanics collections",
            "notes": "QM7, QM7b, QM8, and QM9.",
            "reporting_status": "reported"
          },
          {
            "basis": "original paper dataset collections",
            "count": 3,
            "exclusive": true,
            "exhaustive": true,
            "id": "moleculenet-physical",
            "label": "Physical chemistry collections",
            "notes": "ESOL, FreeSolv, and Lipophilicity.",
            "reporting_status": "reported"
          },
          {
            "basis": "original paper dataset collections",
            "count": 5,
            "exclusive": true,
            "exhaustive": true,
            "id": "moleculenet-biophysics",
            "label": "Biophysics collections",
            "notes": "PCBA, MUV, HIV, BACE, and PDBbind.",
            "reporting_status": "reported"
          },
          {
            "basis": "original paper dataset collections",
            "count": 5,
            "exclusive": true,
            "exhaustive": true,
            "id": "moleculenet-physiology",
            "label": "Physiology collections",
            "notes": "BBBP, Tox21, ToxCast, SIDER, and ClinTox.",
            "reporting_status": "reported"
          }
        ],
        "total": 17
      },
      "task_formats": [
        "molecular property regression",
        "molecular property classification",
        "multitask endpoint prediction"
      ],
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Creator paper Table 1 defines the fixed collection, recommended splits, and metrics.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": null,
          "evidence_ids": [
            "moleculenet-paper-definition-evidence",
            "moleculenet-repository-evidence"
          ],
          "formal_tracks": [],
          "id": "moleculenet-original-2017",
          "label": "original-2017",
          "notes": "Fixed original-paper snapshot of 17 collections; later additions to DeepChem MolNet are outside this version.",
          "release_date": "2017-03-02",
          "status": "current",
          "task_counts": {
            "basis": "original paper dataset collections",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "original paper dataset collections",
                "count": 4,
                "exclusive": true,
                "exhaustive": true,
                "id": "moleculenet-quantum",
                "label": "Quantum mechanics collections",
                "notes": "QM7, QM7b, QM8, and QM9.",
                "reporting_status": "reported"
              },
              {
                "basis": "original paper dataset collections",
                "count": 3,
                "exclusive": true,
                "exhaustive": true,
                "id": "moleculenet-physical",
                "label": "Physical chemistry collections",
                "notes": "ESOL, FreeSolv, and Lipophilicity.",
                "reporting_status": "reported"
              },
              {
                "basis": "original paper dataset collections",
                "count": 5,
                "exclusive": true,
                "exhaustive": true,
                "id": "moleculenet-biophysics",
                "label": "Biophysics collections",
                "notes": "PCBA, MUV, HIV, BACE, and PDBbind.",
                "reporting_status": "reported"
              },
              {
                "basis": "original paper dataset collections",
                "count": 5,
                "exclusive": true,
                "exhaustive": true,
                "id": "moleculenet-physiology",
                "label": "Physiology collections",
                "notes": "BBBP, Tox21, ToxCast, SIDER, and ClinTox.",
                "reporting_status": "reported"
              }
            ],
            "total": 17
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "See the linked official creator resources.",
        "biosafety_notes": null,
        "grader": "Not established by the independently verified metadata claims.",
        "level": "fully-open",
        "license": null,
        "tasks": "See the linked official creator resources."
      },
      "aliases": [],
      "audit": {
        "audited_date": "2026-07-31",
        "notes": "Automated double-pass extraction plus deterministic official-resource pin; the benchmark license remains unverified and is published as null.",
        "status": "audited-with-caveats",
        "unresolved_fields": 2
      },
      "benchmark_use_ids": [
        "an-integrative-approach-to-protein-sequence-design-thr-papd-benchmark-3-use",
        "an-integrative-approach-to-protein-sequence-design-thr-papd-benchmark-4-use"
      ],
      "capabilities": [
        "design",
        "optimization"
      ],
      "coverage_notes": [],
      "domains": [
        "life-science",
        "protein-science",
        "protein-sequence",
        "protein-structure",
        "protein-design",
        "protein-protein-binding"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "an-integrative-approach-to-protein-sequence-design-thr"
      ],
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-31",
          "id": "papd-benchmark-automated-metadata-1-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "cbdee6951e5532b57429398b51436fe8e2abfc9f78df018593e82380271ae3fc",
            "type": "section",
            "value": "Data Availability"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/access/level"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "papd-benchmark-automated-metadata-2-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "fb77bf96dd1177ced1e894036d1f38021a551bc2d2586536f0957b1fe6c1c5c7",
            "type": "section",
            "value": "Results — Genetic algorithms can be applied to higher-dimensional design problems"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/capabilities",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "papd-benchmark-automated-metadata-3-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "c3fa14ff4a216ab69805c6f2078d528a9dfbee0e37ee6372d78e27e001e36b8e",
            "type": "section",
            "value": "Results — Genetic algorithms can be applied to higher-dimensional design problems"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/domains"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "papd-benchmark-automated-metadata-4-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "6e6903443229eedd52871ff2829f1ae65202e38c28b6be6dcf62d0c145ae0384",
            "type": "section",
            "value": "Data Availability"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/kind"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "papd-benchmark-automated-metadata-5-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "9e679f94db5ad08134b7b062f8a1c29b3afb20cfae96940036dedce90a756bd1",
            "type": "section",
            "value": "Results — Genetic algorithms can be applied to higher-dimensional design problems"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/modalities"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "papd-benchmark-automated-metadata-6-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "848cd33a2bc2f062912f7f517b0dd4422bf797fb25a4a1a0eea407983c427d30",
            "type": "section",
            "value": "Methods — Structure preparation"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/name"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "papd-benchmark-automated-metadata-7-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "06c4d9bdb31c2ac3c4e3ceb12dff9a706e3bc8b0d242cbef9e52e7f81cb2a18e",
            "type": "other",
            "value": "Front matter — author-affiliation mapping"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/organizations"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "papd-benchmark-automated-metadata-8-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "0c14f8d38b38f7f6d49953960b0792f82518e785b9d7dd19a3554714df109818",
            "type": "section",
            "value": "Results — Genetic algorithms can be applied to higher-dimensional design problems"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/summary"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "papd-benchmark-automated-bibliographic-evidence",
          "locator": {
            "document_page": null,
            "note": "Resolved from the canonical paper identifier during intake.",
            "printed_page": null,
            "source_fragment_sha256": "5b45963847a151050eb2419c23684ca38e9325e2457145144b7ec0d2ea01d410",
            "type": "other",
            "value": "Crossref bibliographic metadata"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/release_date"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "papd-benchmark-automated-count-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "c4fc0e4b4019757c5cfba59423f2766c466fc9f993eec8de7ee759b14fe80b65",
            "type": "figure",
            "value": "Fig. 1"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/task_counts/total",
            "/task_counts/basis",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "papd-benchmark-automated-subset-coverage-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "c4fc0e4b4019757c5cfba59423f2766c466fc9f993eec8de7ee759b14fe80b65",
            "type": "figure",
            "value": "Fig. 1"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/task_counts/subsets",
            "/versions/0/task_counts/subsets"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "papd-benchmark-automated-resource-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "cbdee6951e5532b57429398b51436fe8e2abfc9f78df018593e82380271ae3fc",
            "type": "section",
            "value": "Data Availability"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/resources",
            "/implementations",
            "/access/license"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "papd-benchmark-automated-creator-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "6115205069f1a4999f060235c61c133f276780408241486c68044ff841a14049",
            "type": "other",
            "value": "Front matter — DOI"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/resources/0"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "papd-benchmark-automated-version-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "6115205069f1a4999f060235c61c133f276780408241486c68044ff841a14049",
            "type": "other",
            "value": "Front matter — DOI"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/latest_version",
            "/versions/0"
          ]
        }
      ],
      "field_status": [
        {
          "confidence": "high",
          "evidence_ids": [
            "papd-benchmark-automated-subset-coverage-evidence"
          ],
          "path": "/task_counts/subsets",
          "reason": "The double-pass review established the root item total but did not establish an exhaustive formal-subset inventory; the empty list must not be interpreted as evidence that the benchmark has no subsets.",
          "status": "provisional"
        },
        {
          "confidence": "high",
          "evidence_ids": [
            "papd-benchmark-automated-resource-evidence"
          ],
          "path": "/access/license",
          "reason": "The double-pass review verified the official resource identity but did not establish a redistributable benchmark license; the value remains null pending source-level license verification.",
          "status": "provisional"
        }
      ],
      "id": "papd-benchmark",
      "implementations": [
        {
          "commit": "b16a0ef3d5c44f65714b1a6c51826f6b4bdaa998",
          "framework": "official repository",
          "notes": "Commit resolved deterministically during paper intake.",
          "status": "official",
          "url": "https://github.com/luhong88/int_seq_des"
        }
      ],
      "kind": "dataset",
      "latest_version": "initial-release",
      "modalities": [
        "protein-sequence",
        "structure-3d"
      ],
      "name": "PapD benchmark",
      "organizations": [
        "University of California, San Francisco",
        "Quantitative Biosciences Institute",
        "Chan Zuckerberg Biohub"
      ],
      "parent_id": null,
      "release_date": "2024-07-11",
      "resources": [
        {
          "access_notes": "Versioned creator source; full text is not mirrored.",
          "id": "papd-benchmark-creator-paper-resource",
          "last_checked": "2026-07-31",
          "license": null,
          "pin": null,
          "type": "paper",
          "url": "https://doi.org/10.1371/journal.pcbi.1011953"
        },
        {
          "access_notes": "Official resource pinned immutably during intake.",
          "id": "papd-benchmark-official-repository-resource",
          "last_checked": "2026-07-31",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://github.com/luhong88/int_seq_des/commit/b16a0ef3d5c44f65714b1a6c51826f6b4bdaa998",
            "value": "b16a0ef3d5c44f65714b1a6c51826f6b4bdaa998"
          },
          "type": "repository",
          "url": "https://github.com/luhong88/int_seq_des"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "initial-release",
        "entries": [
          {
            "confidence": "high",
            "count": null,
            "count_basis": "Creator-defined multistate protein sequence-design benchmark system.",
            "count_ref": null,
            "count_unit": "systems",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "papd-benchmark-automated-metadata-2-evidence"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "The three modeled binding states are objective dimensions, not three independent sequence-design tasks.",
            "reporting_status": "not_reported",
            "task_type_id": "protein-sequence-design"
          }
        ],
        "notes": "The creator paper explicitly frames this benchmark system as multistate protein sequence design; more specific downstream design objectives are not exhaustively classified.",
        "status": "partial"
      },
      "summary": "A multistate protein sequence-design benchmark targeting the multispecific PapD binding interface.",
      "task_counts": {
        "basis": "Fig. 1B explicitly defines the complete PapD design problem as three-state.",
        "reporting_status": "reported",
        "subsets": [],
        "total": 3
      },
      "task_formats": [
        "unclassified"
      ],
      "verification": {
        "last_verified": "2026-07-31",
        "notes": "New family admitted after creator source and official resource verification; the unresolved benchmark license is visibly flagged.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": null,
          "evidence_ids": [
            "papd-benchmark-automated-count-evidence",
            "papd-benchmark-automated-version-evidence"
          ],
          "formal_tracks": [],
          "id": "papd-benchmark-initial-release-version",
          "label": "initial-release",
          "notes": "Atlas snapshot label for the creator-paper release; the source does not report a formal benchmark version.",
          "release_date": "2024-07-11",
          "status": "current",
          "task_counts": {
            "basis": "Fig. 1B explicitly defines the complete PapD design problem as three-state.",
            "reporting_status": "reported",
            "subsets": [],
            "total": 3
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "Excel annotation spreadsheet and PDB complex-structure files are available from the linked Zenodo record.",
        "biosafety_notes": null,
        "grader": "Not established by the independently verified metadata claims.",
        "level": "fully-open",
        "license": "CC BY 4.0",
        "tasks": "See the linked official creator resources."
      },
      "aliases": [],
      "audit": {
        "audited_date": "2026-07-29",
        "notes": "Automated double-pass extraction plus deterministic repository pin. Owner review preserved the corroborated root total and excluded conflicted appendix inventory subcounts.",
        "status": "audited-with-caveats",
        "unresolved_fields": 2
      },
      "benchmark_use_ids": [
        "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-1-use",
        "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-2-use",
        "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-3-use",
        "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-4-use",
        "ppb-affinity-protein-protein-binding-affinity-dataset-ppb-affinity-1-use",
        "ppb-affinity-protein-protein-binding-affinity-dataset-ppb-affinity-2-use",
        "ppb-affinity-protein-protein-binding-affinity-dataset-ppb-affinity-3-use"
      ],
      "capabilities": [
        "prediction"
      ],
      "coverage_notes": [],
      "domains": [
        "protein-protein-binding"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "explainable-protein-protein-binding-affinity-predictio",
        "ppb-affinity-protein-protein-binding-affinity-dataset"
      ],
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-29",
          "id": "ppb-affinity-automated-metadata-1-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "b8bf85cbf0204d3d141587041a562e18f899405823449dfd01c2fb0b97d3d913",
            "type": "section",
            "value": "Data Record, paragraph 24"
          },
          "source_id": "ppb-affinity-protein-protein-binding-affinity-dataset",
          "source_type": "work",
          "supports": [
            "/access/artifacts"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "ppb-affinity-automated-metadata-2-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "1450e68765093017485978d55be5219857370e490aaadc31533870e83f6da55e",
            "type": "section",
            "value": "Data Record, paragraph 24; Usage Notes, paragraph 53"
          },
          "source_id": "ppb-affinity-protein-protein-binding-affinity-dataset",
          "source_type": "work",
          "supports": [
            "/access/level"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "ppb-affinity-automated-metadata-3-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "8d75fb5e76c2de189a8740a03c26b1120716dee92b2f959428e0b808c1880500",
            "type": "section",
            "value": "Data Record, paragraph 24"
          },
          "source_id": "ppb-affinity-protein-protein-binding-affinity-dataset",
          "source_type": "work",
          "supports": [
            "/access/license"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "ppb-affinity-automated-metadata-4-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "8723c300a175ea59a647ad3378ea8af168e44a14a3dfbdce2f192805a0c6256e",
            "type": "section",
            "value": "Usage Notes > Potential uses of the dataset, paragraph 54"
          },
          "source_id": "ppb-affinity-protein-protein-binding-affinity-dataset",
          "source_type": "work",
          "supports": [
            "/capabilities"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "ppb-affinity-automated-metadata-5-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "7ee57c07ac310ba4643b7bf61b9f5b08c147d707b5ffccbf2371084aaee5e4e1",
            "type": "section",
            "value": "Abstract, paragraph 1"
          },
          "source_id": "ppb-affinity-protein-protein-binding-affinity-dataset",
          "source_type": "work",
          "supports": [
            "/domains"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "ppb-affinity-automated-metadata-6-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "48c0f429f515849a116d1501c9dad36b693a3330fde8132c803cff68af53a919",
            "type": "section",
            "value": "Abstract, paragraph 1"
          },
          "source_id": "ppb-affinity-protein-protein-binding-affinity-dataset",
          "source_type": "work",
          "supports": [
            "/kind"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "ppb-affinity-automated-metadata-7-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "26722d095e1636353ef0a3294d8b797a99a76b2db39e1d9f7dba831fcba21a5b",
            "type": "section",
            "value": "Data Record, paragraph 24"
          },
          "source_id": "ppb-affinity-protein-protein-binding-affinity-dataset",
          "source_type": "work",
          "supports": [
            "/modalities"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "ppb-affinity-automated-metadata-8-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "45b00c34fea2a43873b0a36fcb8dd2f3963a44228c7b06d9f2b95ec8ed23a93b",
            "type": "section",
            "value": "Background & Summary, paragraph 4"
          },
          "source_id": "ppb-affinity-protein-protein-binding-affinity-dataset",
          "source_type": "work",
          "supports": [
            "/name"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "ppb-affinity-automated-metadata-9-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "c4cc5b489463659307019f93038ba4ca743acb43294d201882bc264f99f4364d",
            "type": "section",
            "value": "Front matter > Author affiliations"
          },
          "source_id": "ppb-affinity-protein-protein-binding-affinity-dataset",
          "source_type": "work",
          "supports": [
            "/organizations"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "ppb-affinity-automated-metadata-10-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "85eb31244fc784ed22a17b58148e6f6349d8f6c2659130b6faaca9af3edd13f6",
            "type": "section",
            "value": "Methods > Data processing and quality assessment, paragraph 6; Data Record, paragraph 24"
          },
          "source_id": "ppb-affinity-protein-protein-binding-affinity-dataset",
          "source_type": "work",
          "supports": [
            "/summary"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "ppb-affinity-automated-bibliographic-evidence",
          "locator": {
            "document_page": null,
            "note": "Resolved from the canonical paper identifier during intake.",
            "printed_page": null,
            "source_fragment_sha256": "671a1aa1f690e1304ea98c4138eb16fe0b599317c244c30c2a6c49fede0b57c5",
            "type": "other",
            "value": "Crossref bibliographic metadata"
          },
          "source_id": "ppb-affinity-protein-protein-binding-affinity-dataset",
          "source_type": "work",
          "supports": [
            "/release_date"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "ppb-affinity-automated-count-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "942ae775b976de7b1842df66c8a6fa69a5d57e5a88f313ac60a40237c4d36d40",
            "type": "section",
            "value": "Methods > Data processing and quality assessment > Protein chains, paragraphs 16–18; Data Record > Metadata, paragraph 43"
          },
          "source_id": "ppb-affinity-protein-protein-binding-affinity-dataset",
          "source_type": "work",
          "supports": [
            "/task_counts/total",
            "/versions/0/task_counts"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "ppb-affinity-automated-resource-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "6ec0069c26ea51711ae09252a86c7b36446fb05edcada02090deb505983eb6a6",
            "type": "section",
            "value": "Code availability"
          },
          "source_id": "ppb-affinity-protein-protein-binding-affinity-dataset",
          "source_type": "work",
          "supports": [
            "/resources",
            "/implementations"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "ppb-affinity-automated-creator-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "ce6add923fbd5cf360b0269cfee4408ebadcb46335186012664e7bcdecc0b23c",
            "type": "section",
            "value": "Data Record, paragraph 24"
          },
          "source_id": "ppb-affinity-protein-protein-binding-affinity-dataset",
          "source_type": "work",
          "supports": [
            "/resources/0"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "ppb-affinity-automated-version-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "ce6add923fbd5cf360b0269cfee4408ebadcb46335186012664e7bcdecc0b23c",
            "type": "section",
            "value": "Data Record, paragraph 24"
          },
          "source_id": "ppb-affinity-protein-protein-binding-affinity-dataset",
          "source_type": "work",
          "supports": [
            "/latest_version",
            "/versions/0"
          ]
        },
        {
          "accessed_date": "2026-07-29",
          "id": "ppb-affinity-automated-count-conflict-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "942ae775b976de7b1842df66c8a6fa69a5d57e5a88f313ac60a40237c4d36d40",
            "type": "section",
            "value": "Methods > Data processing and quality assessment > Protein chains, paragraphs 16–18; Data Record > Metadata, paragraph 43"
          },
          "source_id": "ppb-affinity-protein-protein-binding-affinity-dataset",
          "source_type": "work",
          "supports": [
            "/task_counts/basis",
            "/task_counts/subsets"
          ]
        }
      ],
      "field_status": [
        {
          "confidence": "high",
          "evidence_ids": [
            "ppb-affinity-automated-count-conflict-evidence"
          ],
          "path": "/task_counts/basis",
          "reason": "The owner approved the explicit root-total value after the extractor reported it and the verifier independently located it at high confidence; the detailed uniqueness basis remains conflicted and all subcounts are excluded from publication.",
          "status": "conflicted"
        },
        {
          "confidence": "high",
          "evidence_ids": [
            "ppb-affinity-automated-count-conflict-evidence"
          ],
          "path": "/task_counts/subsets",
          "reason": "The owner approved the explicit root-total value after the extractor reported it and the verifier independently located it at high confidence; the detailed uniqueness basis remains conflicted and all subcounts are excluded from publication.",
          "status": "conflicted"
        }
      ],
      "id": "ppb-affinity",
      "implementations": [
        {
          "commit": "f1a1698ba868365af8a55807f8ae5b0d9653fa97",
          "framework": "official repository",
          "notes": "Commit resolved deterministically during paper intake.",
          "status": "official",
          "url": "https://github.com/Huatsing-Lau/PPB-Affinity-DataPrepWorkflow"
        }
      ],
      "kind": "dataset",
      "latest_version": "initial-release",
      "modalities": [
        "table",
        "structure-3d"
      ],
      "name": "PPB-Affinity",
      "organizations": [
        "Artificial Intelligence Innovation Center, Research Institute of Tsinghua, Pearl River Delta",
        "Cyagen Biosciences (Suzhou) Inc.",
        "Cyagen Biosciences (Guangzhou) Inc.",
        "Cyagen Biomodels (Guangzhou) Co., Ltd",
        "Department of Pain Medicine, Shenzhen Nanshan People’s Hospital, Shenzhen University Medical School"
      ],
      "parent_id": null,
      "release_date": "2024-12-03",
      "resources": [
        {
          "access_notes": "Versioned creator source; full text is not mirrored.",
          "id": "ppb-affinity-creator-paper-resource",
          "last_checked": "2026-07-29",
          "license": null,
          "pin": null,
          "type": "paper",
          "url": "https://zenodo.org/doi/10.5281/zenodo.11070823"
        },
        {
          "access_notes": "Official repository pinned during intake.",
          "id": "ppb-affinity-official-repository-resource",
          "last_checked": "2026-07-29",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://github.com/Huatsing-Lau/PPB-Affinity-DataPrepWorkflow/commit/f1a1698ba868365af8a55807f8ae5b0d9653fa97",
            "value": "f1a1698ba868365af8a55807f8ae5b0d9653fa97"
          },
          "type": "repository",
          "url": "https://github.com/Huatsing-Lau/PPB-Affinity-DataPrepWorkflow"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "initial-release",
        "entries": [],
        "notes": "No Scientific Task claim passed independent high-confidence verification; task mapping remains pending a targeted official-source audit.",
        "status": "partial"
      },
      "summary": "A reusable protein-protein binding-affinity dataset with complex structures, measured affinities, receptor and ligand chains, and mutation annotations.",
      "task_counts": {
        "basis": "Whole-dataset unique-sample total explicitly reported by the source; the detailed uniqueness basis remains conflicted.",
        "reporting_status": "reported",
        "subsets": [],
        "total": 12062
      },
      "task_formats": [
        "unclassified"
      ],
      "verification": {
        "last_verified": "2026-07-29",
        "notes": "New family admitted with an explicit count-inventory caveat after creator source, official repository, double-pass verification, and owner conflict resolution.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": null,
          "evidence_ids": [
            "ppb-affinity-automated-count-evidence",
            "ppb-affinity-automated-version-evidence"
          ],
          "formal_tracks": [],
          "id": "ppb-affinity-initial-release-version",
          "label": "initial-release",
          "notes": "Root total retained after owner review; conflicted appendix inventory subcounts are intentionally omitted.",
          "release_date": "2024-12-03",
          "status": "current",
          "task_counts": {
            "basis": "Whole-dataset unique-sample total explicitly reported by the source; the detailed uniqueness basis remains conflicted.",
            "reporting_status": "reported",
            "subsets": [],
            "total": 12062
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "Official releases include reference files, model scores, MSAs, predicted structures, cross-validation folds, and raw/processed benchmark data.",
        "biosafety_notes": "The creator paper explicitly discusses dual-use risk for protein fitness and design models, including harmful optimization; this registry mirrors metadata and links only.",
        "grader": "Public deterministic scoring scripts compute the official DMS and clinical metrics.",
        "level": "fully-open",
        "license": "MIT",
        "tasks": "Processed DMS assay tables and clinical benchmark files are publicly downloadable in versioned archives."
      },
      "aliases": [
        "Protein Gym"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Versioned release archives take precedence for task counts. The current indel-assay count remains marked conflicted because the v1.3 archive/reference file contains 66 assays while the pinned official README says 74.",
        "status": "audited-with-caveats",
        "unresolved_fields": 2
      },
      "benchmark_use_ids": [
        "system-card-claude-opus-5-proteingym-4-use"
      ],
      "capabilities": [
        "prediction",
        "regression",
        "classification",
        "design",
        "optimization"
      ],
      "coverage_notes": [
        {
          "count": null,
          "coverage": "explicitly-in-scope",
          "notes": "NDCG@10% and top-10% recall evaluate design-oriented ranking, but design is not a separately counted task subset.",
          "reporting_status": "not_reported",
          "tag": "protein-design"
        },
        {
          "count": null,
          "coverage": "explicitly-in-scope",
          "notes": "The paper explicitly includes ligand binding and reports a generic Binding function category: 14 substitution assays in v1.0 and 13 in v1.1-v1.3. It does not publish a target-type split, so no protein-ligand-only count is inferred.",
          "reporting_status": "not_reported",
          "tag": "protein-ligand-binding"
        },
        {
          "count": null,
          "coverage": "observed",
          "notes": "Some source assays concern protein-protein binding, but the official function category is generic Binding and no protein-protein-only count is published.",
          "reporting_status": "not_reported",
          "tag": "protein-protein-binding"
        }
      ],
      "domains": [
        "protein-sequence",
        "protein-design",
        "protein-ligand-binding",
        "clinical-translational"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "system-card-claude-opus-5"
      ],
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "proteingym-evidence-identity",
          "locator": {
            "note": "Official identity, creator affiliations, benchmark scope, modalities, task formats, and evaluation regimes.",
            "type": "page",
            "value": "pp. 1–6, title, author affiliations, abstract, Figure 1, and Sections 3–4"
          },
          "source_id": "proteingym-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/kind",
            "/summary",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "proteingym-evidence-taxonomy",
          "locator": {
            "note": "DMS/clinical regimes, design-oriented metrics, ligand-binding scope, and the generic Binding function category.",
            "type": "table",
            "value": "pp. 5–6 and 30, Table 1, Section 4.1, and Table A2"
          },
          "source_id": "proteingym-paper",
          "source_type": "work",
          "supports": [
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/coverage_notes",
            "/scientific_task_classification/entries/0",
            "/scientific_task_classification/entries/1",
            "/scientific_task_classification/entries/2",
            "/scientific_task_classification/entries/3"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "proteingym-evidence-v10-counts",
          "locator": {
            "note": "v1.0 reports 217 substitution assays, 66 indel assays, 2,525/1,555 clinical proteins, and 14 generic Binding substitution assays.",
            "type": "table",
            "value": "pp. 6, 29–30; Tables 1, A1, and A2"
          },
          "source_id": "proteingym-paper",
          "source_type": "work",
          "supports": [
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "proteingym-evidence-v10-release",
          "locator": {
            "note": "The benchmark release date is anchored to the creator preprint posted on 2023-12-08 because the dataset record only reports month precision.",
            "type": "dataset-card",
            "value": "Zenodo record 13932633 metadata: version 1.0, publication date 2023-12, open access, MIT license"
          },
          "source_id": "proteingym-v10-dataset-resource",
          "source_type": "resource",
          "supports": [
            "/release_date",
            "/versions/0/release_date",
            "/resources"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "proteingym-evidence-v11-counts",
          "locator": {
            "note": "CSV row counts are 217, 66, 2,525, and 1,555; coarse_selection_type has 13 Binding rows.",
            "type": "repository-path",
            "value": "ProteinGym_v1.1.zip: DMS_substitutions.csv, DMS_indels.csv, clinical_substitutions.csv, clinical_indels.csv"
          },
          "source_id": "proteingym-v11-dataset-resource",
          "source_type": "resource",
          "supports": [
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "proteingym-evidence-v12-counts",
          "locator": {
            "note": "CSV row counts are 217, 66, 2,525, and 1,555; coarse_selection_type has 13 Binding rows.",
            "type": "repository-path",
            "value": "Zenodo record 14997691: DMS_substitutions.csv, DMS_indels.csv, clinical_substitutions.csv, clinical_indels.csv"
          },
          "source_id": "proteingym-v12-dataset-resource",
          "source_type": "resource",
          "supports": [
            "/versions/2/task_counts/total",
            "/versions/2/task_counts/basis",
            "/versions/2/task_counts/subsets"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "proteingym-evidence-v13-counts",
          "locator": {
            "note": "Reference/archive row counts are 217, 66, 2,525, and 1,555; coarse_selection_type has 13 Binding rows.",
            "type": "repository-path",
            "value": "Zenodo record 15293562: DMS_substitutions.csv, DMS_indels.csv, clinical_substitutions.csv, clinical_indels.csv and DMS_ProteinGym_indels.zip"
          },
          "source_id": "proteingym-v13-dataset-resource",
          "source_type": "resource",
          "supports": [
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/task_counts/subsets/1/count",
            "/versions/3/task_counts/total",
            "/versions/3/task_counts/basis",
            "/versions/3/task_counts/subsets",
            "/versions/3/task_counts/subsets/1/count"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "proteingym-evidence-releases",
          "locator": {
            "note": "The README identifies v1.3 as latest and links all four immutable Zenodo releases.",
            "type": "release",
            "value": "PG_v1.0–PG_v1.3 tags and README Releases section at commit 1f8de974dead8ff7501eff087b725d14a965e9f9"
          },
          "source_id": "proteingym-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/versions/1/release_date",
            "/versions/2/release_date",
            "/versions/3/release_date",
            "/resources",
            "/implementations"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "proteingym-evidence-current-readme",
          "locator": {
            "note": "README reports 217 DMS substitution assays, 74 DMS indel assays, 2,525/1,555 clinical proteins, metrics, aggregation, downloads, and MIT license.",
            "type": "repository-path",
            "value": "README.md, Overview and Results at tag PG_v1.3"
          },
          "source_id": "proteingym-repository-resource",
          "source_type": "resource",
          "supports": [
            "/task_counts/subsets/1/count",
            "/versions/3/task_counts/subsets/1/count",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/implementations"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "proteingym-evidence-access-safety",
          "locator": {
            "note": "Public code/data/model-score resources, MIT code license, and explicit dual-use discussion.",
            "type": "section",
            "value": "pp. 11 and 28–29, Section 6 and Appendices A.1 and A.3.3–A.3.4"
          },
          "source_id": "proteingym-paper",
          "source_type": "work",
          "supports": [
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/resources",
            "/implementations"
          ]
        }
      ],
      "field_status": [
        {
          "confidence": "medium",
          "evidence_ids": [
            "proteingym-evidence-v13-counts",
            "proteingym-evidence-current-readme"
          ],
          "path": "/task_counts/subsets/1/count",
          "reason": "The versioned v1.3 archive contains 66 DMS indel assay records, while the official PG_v1.3 README states 74. The archive value is retained under the source-priority policy.",
          "status": "conflicted"
        },
        {
          "confidence": "medium",
          "evidence_ids": [
            "proteingym-evidence-v13-counts",
            "proteingym-evidence-current-readme"
          ],
          "path": "/versions/3/task_counts/subsets/1/count",
          "reason": "The versioned v1.3 archive contains 66 DMS indel assay records, while the official PG_v1.3 README states 74. The archive value is retained under the source-priority policy.",
          "status": "conflicted"
        }
      ],
      "id": "proteingym",
      "implementations": [
        {
          "commit": "1f8de974dead8ff7501eff087b725d14a965e9f9",
          "framework": "ProteinGym scoring pipeline",
          "notes": "Commit referenced by the official PG_v1.3 tag; scoring scripts are baseline- and track-specific.",
          "status": "official",
          "url": "https://github.com/OATML-Markslab/ProteinGym/tree/1f8de974dead8ff7501eff087b725d14a965e9f9"
        }
      ],
      "kind": "suite",
      "latest_version": "1.3",
      "modalities": [
        "protein-sequence",
        "structure-3d",
        "table"
      ],
      "name": "ProteinGym",
      "organizations": [
        "University of Oxford",
        "Harvard Medical School",
        "Seismic Therapeutic",
        "Harvard University",
        "Centre for Genomic Regulation",
        "Universitat Pompeu Fabra",
        "Broad Institute"
      ],
      "parent_id": null,
      "release_date": "2023-12-08",
      "resources": [
        {
          "access_notes": "Final NeurIPS 2023 Datasets and Benchmarks paper for ProteinGym v1.0.",
          "id": "proteingym-paper-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "snapshot",
            "url": "https://proceedings.neurips.cc/paper_files/paper/2023/file/cac723e5ff29f65e3fcbb0739ae91bee-Paper-Datasets_and_Benchmarks.pdf",
            "value": "sha256:a3b08cc4a6befd64620cf0f287d78d55a36dc2639955f52c5655f83833c50104"
          },
          "type": "paper",
          "url": "https://proceedings.neurips.cc/paper_files/paper/2023/file/cac723e5ff29f65e3fcbb0739ae91bee-Paper-Datasets_and_Benchmarks.pdf"
        },
        {
          "access_notes": "Official living leaderboards and per-assay performance views.",
          "id": "proteingym-website-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": null,
          "type": "website",
          "url": "https://www.proteingym.org/"
        },
        {
          "access_notes": "Official code, release links, reference files, model scores, and scoring pipeline.",
          "id": "proteingym-repository-resource",
          "last_checked": "2026-07-21",
          "license": "MIT",
          "pin": {
            "kind": "tag",
            "url": "https://github.com/OATML-Markslab/ProteinGym/tree/1f8de974dead8ff7501eff087b725d14a965e9f9",
            "value": "PG_v1.3"
          },
          "type": "repository",
          "url": "https://github.com/OATML-Markslab/ProteinGym"
        },
        {
          "access_notes": "Official versioned v1.0 archive.",
          "id": "proteingym-v10-dataset-resource",
          "last_checked": "2026-07-21",
          "license": "MIT",
          "pin": {
            "kind": "version",
            "url": "https://zenodo.org/records/13932633",
            "value": "1.0"
          },
          "type": "dataset",
          "url": "https://doi.org/10.5281/zenodo.13932633"
        },
        {
          "access_notes": "Official versioned v1.1 archive.",
          "id": "proteingym-v11-dataset-resource",
          "last_checked": "2026-07-21",
          "license": "MIT",
          "pin": {
            "kind": "version",
            "url": "https://zenodo.org/records/13936340",
            "value": "1.1"
          },
          "type": "dataset",
          "url": "https://doi.org/10.5281/zenodo.13936340"
        },
        {
          "access_notes": "Official versioned v1.2 archive.",
          "id": "proteingym-v12-dataset-resource",
          "last_checked": "2026-07-21",
          "license": "MIT",
          "pin": {
            "kind": "version",
            "url": "https://zenodo.org/records/14997691",
            "value": "1.2"
          },
          "type": "dataset",
          "url": "https://doi.org/10.5281/zenodo.14997691"
        },
        {
          "access_notes": "Official versioned v1.3 archive and current registry snapshot.",
          "id": "proteingym-v13-dataset-resource",
          "last_checked": "2026-07-21",
          "license": "MIT",
          "pin": {
            "kind": "version",
            "url": "https://zenodo.org/records/15293562",
            "value": "1.3"
          },
          "type": "dataset",
          "url": "https://doi.org/10.5281/zenodo.15293562"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "1.3",
        "entries": [
          {
            "confidence": "high",
            "count": null,
            "count_basis": "DMS assays and mutant measurements across version 1.3 tracks.",
            "count_ref": null,
            "count_unit": "assays",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "proteingym-evidence-taxonomy"
            ],
            "mapping_method": "official-track",
            "notes": "Track-specific assay counts are recorded on child records.",
            "reporting_status": "not_reported",
            "task_type_id": "protein-mutation-effect-prediction"
          },
          {
            "confidence": "high",
            "count": null,
            "count_basis": "ProteinGym DMS assays.",
            "count_ref": null,
            "count_unit": "assays",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "proteingym-evidence-taxonomy"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "No heterogeneous root total is asserted.",
            "reporting_status": "not_reported",
            "task_type_id": "protein-fitness-prediction"
          },
          {
            "confidence": "high",
            "count": null,
            "count_basis": "Clinical substitution and indel tracks.",
            "count_ref": null,
            "count_unit": "records",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "proteingym-evidence-taxonomy"
            ],
            "mapping_method": "official-track",
            "notes": "Child records use clinical proteins as their count unit.",
            "reporting_status": "not_reported",
            "task_type_id": "protein-clinical-variant-interpretation"
          },
          {
            "confidence": "high",
            "count": null,
            "count_basis": "DMS assays in the official generic Binding function category.",
            "count_ref": null,
            "count_unit": "assays",
            "coverage": "observed",
            "evidence_ids": [
              "proteingym-evidence-taxonomy"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "The source does not support a ligand-only or PPI-only assay count.",
            "reporting_status": "not_reported",
            "task_type_id": "protein-ligand-binding-prediction"
          }
        ],
        "notes": "The release mixes DMS assays, clinical proteins, and mutant measurements; counts stay on formal child tracks.",
        "status": "partial"
      },
      "summary": "Versioned deep-mutational-scanning and clinical-variant benchmarks for protein fitness prediction and design in zero-shot and supervised regimes.",
      "task_counts": {
        "basis": "heterogeneous DMS assays, clinical proteins, and mutant measurements",
        "reporting_status": "not_reported",
        "subsets": [
          {
            "basis": "assays",
            "count": 217,
            "exclusive": false,
            "exhaustive": false,
            "id": "dms-substitution-assays",
            "label": "DMS substitution assays",
            "notes": "The v1.3 reference file has 217 assay rows; assay count is distinct from the README description of approximately 2.7 million missense variants.",
            "reporting_status": "reported"
          },
          {
            "basis": "assays",
            "count": 66,
            "exclusive": false,
            "exhaustive": false,
            "id": "dms-indel-assays",
            "label": "DMS indel assays",
            "notes": "The v1.3 versioned archive has 66 assay records; the pinned repository README says 74, so this value is marked Conflicted.",
            "reporting_status": "reported"
          },
          {
            "basis": "proteins",
            "count": 2525,
            "exclusive": false,
            "exhaustive": false,
            "id": "clinical-substitution-proteins",
            "label": "Clinical substitution proteins",
            "notes": "The v1.3 clinical reference file has one row per protein.",
            "reporting_status": "reported"
          },
          {
            "basis": "proteins",
            "count": 1555,
            "exclusive": false,
            "exhaustive": false,
            "id": "clinical-indel-proteins",
            "label": "Clinical indel proteins",
            "notes": "The v1.3 clinical indel reference file has one row per protein.",
            "reporting_status": "reported"
          },
          {
            "basis": "assays",
            "count": 13,
            "exclusive": false,
            "exhaustive": false,
            "id": "dms-substitution-binding-assays",
            "label": "DMS substitution assays in the Binding function category",
            "notes": "The v1.3 reference file contains 13 rows with coarse_selection_type=Binding; target type is not split into protein-protein versus protein-ligand counts.",
            "reporting_status": "reported"
          }
        ],
        "total": null
      },
      "task_formats": [
        "zero-shot mutation-effect prediction",
        "supervised mutation-effect prediction",
        "clinical variant classification"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Field-level audit completed against the final creator paper, official versioned Zenodo archives, official repository tag PG_v1.3, and official website.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2023-12-08",
          "evidence_ids": [
            "proteingym-evidence-v10-counts",
            "proteingym-evidence-v10-release"
          ],
          "formal_tracks": [
            "proteingym-dms-substitutions",
            "proteingym-dms-indels",
            "proteingym-clinical-substitutions",
            "proteingym-clinical-indels"
          ],
          "id": "proteingym-v10",
          "label": "1.0",
          "notes": "Initial creator-paper release; the paper explicitly identifies its benchmark as ProteinGym v1.0.",
          "release_date": "2023-12-08",
          "status": "superseded",
          "task_counts": {
            "basis": "heterogeneous DMS assays, clinical proteins, and mutant measurements",
            "reporting_status": "not_reported",
            "subsets": [
              {
                "basis": "assays",
                "count": 217,
                "exclusive": false,
                "exhaustive": false,
                "id": "dms-substitution-assays",
                "label": "DMS substitution assays",
                "notes": "Table A1 reports 2.4 million substitution mutants.",
                "reporting_status": "reported"
              },
              {
                "basis": "assays",
                "count": 66,
                "exclusive": false,
                "exhaustive": false,
                "id": "dms-indel-assays",
                "label": "DMS indel assays",
                "notes": "Table A1 reports 289,000 indel mutants.",
                "reporting_status": "reported"
              },
              {
                "basis": "proteins",
                "count": 2525,
                "exclusive": false,
                "exhaustive": false,
                "id": "clinical-substitution-proteins",
                "label": "Clinical substitution proteins",
                "notes": "Table A1 reports 63,000 clinical substitution mutants.",
                "reporting_status": "reported"
              },
              {
                "basis": "proteins",
                "count": 1555,
                "exclusive": false,
                "exhaustive": false,
                "id": "clinical-indel-proteins",
                "label": "Clinical indel proteins",
                "notes": "Table A1 reports 3,000 clinical indel mutants.",
                "reporting_status": "reported"
              },
              {
                "basis": "assays",
                "count": 14,
                "exclusive": false,
                "exhaustive": false,
                "id": "dms-substitution-binding-assays",
                "label": "DMS substitution assays in the Binding function category",
                "notes": "Table A2 reports 14 substitution and zero indel assays in the generic Binding category.",
                "reporting_status": "reported"
              }
            ],
            "total": null
          }
        },
        {
          "as_of": "2024-10-15",
          "evidence_ids": [
            "proteingym-evidence-v11-counts",
            "proteingym-evidence-releases"
          ],
          "formal_tracks": [
            "proteingym-dms-substitutions",
            "proteingym-dms-indels",
            "proteingym-clinical-substitutions",
            "proteingym-clinical-indels"
          ],
          "id": "proteingym-v11",
          "label": "1.1",
          "notes": "Official release notes describe reference-file updates and the addition of ProtSSN and SaProt baselines.",
          "release_date": "2024-10-15",
          "status": "superseded",
          "task_counts": {
            "basis": "heterogeneous DMS assays, clinical proteins, and mutant measurements",
            "reporting_status": "not_reported",
            "subsets": [
              {
                "basis": "assays",
                "count": 217,
                "exclusive": false,
                "exhaustive": false,
                "id": "dms-substitution-assays",
                "label": "DMS substitution assays",
                "notes": "Counted from DMS_substitutions.csv in the versioned archive.",
                "reporting_status": "reported"
              },
              {
                "basis": "assays",
                "count": 66,
                "exclusive": false,
                "exhaustive": false,
                "id": "dms-indel-assays",
                "label": "DMS indel assays",
                "notes": "Counted from DMS_indels.csv in the versioned archive.",
                "reporting_status": "reported"
              },
              {
                "basis": "proteins",
                "count": 2525,
                "exclusive": false,
                "exhaustive": false,
                "id": "clinical-substitution-proteins",
                "label": "Clinical substitution proteins",
                "notes": "Counted from clinical_substitutions.csv in the versioned archive.",
                "reporting_status": "reported"
              },
              {
                "basis": "proteins",
                "count": 1555,
                "exclusive": false,
                "exhaustive": false,
                "id": "clinical-indel-proteins",
                "label": "Clinical indel proteins",
                "notes": "Counted from clinical_indels.csv in the versioned archive.",
                "reporting_status": "reported"
              },
              {
                "basis": "assays",
                "count": 13,
                "exclusive": false,
                "exhaustive": false,
                "id": "dms-substitution-binding-assays",
                "label": "DMS substitution assays in the Binding function category",
                "notes": "The reference file has 13 rows with coarse_selection_type=Binding.",
                "reporting_status": "reported"
              }
            ],
            "total": null
          }
        },
        {
          "as_of": "2025-03-10",
          "evidence_ids": [
            "proteingym-evidence-v12-counts",
            "proteingym-evidence-releases"
          ],
          "formal_tracks": [
            "proteingym-dms-substitutions",
            "proteingym-dms-indels",
            "proteingym-clinical-substitutions",
            "proteingym-clinical-indels"
          ],
          "id": "proteingym-v12",
          "label": "1.2",
          "notes": "Official release notes describe added zero-shot substitution baselines and mutation-level supervised predictions.",
          "release_date": "2025-03-10",
          "status": "superseded",
          "task_counts": {
            "basis": "heterogeneous DMS assays, clinical proteins, and mutant measurements",
            "reporting_status": "not_reported",
            "subsets": [
              {
                "basis": "assays",
                "count": 217,
                "exclusive": false,
                "exhaustive": false,
                "id": "dms-substitution-assays",
                "label": "DMS substitution assays",
                "notes": "Counted from the versioned DMS_substitutions.csv reference file.",
                "reporting_status": "reported"
              },
              {
                "basis": "assays",
                "count": 66,
                "exclusive": false,
                "exhaustive": false,
                "id": "dms-indel-assays",
                "label": "DMS indel assays",
                "notes": "Counted from the versioned DMS_indels.csv reference file.",
                "reporting_status": "reported"
              },
              {
                "basis": "proteins",
                "count": 2525,
                "exclusive": false,
                "exhaustive": false,
                "id": "clinical-substitution-proteins",
                "label": "Clinical substitution proteins",
                "notes": "Counted from the versioned clinical_substitutions.csv reference file.",
                "reporting_status": "reported"
              },
              {
                "basis": "proteins",
                "count": 1555,
                "exclusive": false,
                "exhaustive": false,
                "id": "clinical-indel-proteins",
                "label": "Clinical indel proteins",
                "notes": "Counted from the versioned clinical_indels.csv reference file.",
                "reporting_status": "reported"
              },
              {
                "basis": "assays",
                "count": 13,
                "exclusive": false,
                "exhaustive": false,
                "id": "dms-substitution-binding-assays",
                "label": "DMS substitution assays in the Binding function category",
                "notes": "The reference file has 13 rows with coarse_selection_type=Binding.",
                "reporting_status": "reported"
              }
            ],
            "total": null
          }
        },
        {
          "as_of": "2025-04-27",
          "evidence_ids": [
            "proteingym-evidence-v13-counts",
            "proteingym-evidence-releases",
            "proteingym-evidence-current-readme"
          ],
          "formal_tracks": [
            "proteingym-dms-substitutions",
            "proteingym-dms-indels",
            "proteingym-clinical-substitutions",
            "proteingym-clinical-indels"
          ],
          "id": "proteingym-v13",
          "label": "1.3",
          "notes": "Current official dataset release; release notes add 16 zero-shot substitution baselines without announcing a benchmark-data expansion.",
          "release_date": "2025-04-27",
          "status": "current",
          "task_counts": {
            "basis": "heterogeneous DMS assays, clinical proteins, and mutant measurements",
            "reporting_status": "not_reported",
            "subsets": [
              {
                "basis": "assays",
                "count": 217,
                "exclusive": false,
                "exhaustive": false,
                "id": "dms-substitution-assays",
                "label": "DMS substitution assays",
                "notes": "The v1.3 reference file has 217 assay rows; assay count is distinct from the README description of approximately 2.7 million missense variants.",
                "reporting_status": "reported"
              },
              {
                "basis": "assays",
                "count": 66,
                "exclusive": false,
                "exhaustive": false,
                "id": "dms-indel-assays",
                "label": "DMS indel assays",
                "notes": "The v1.3 versioned archive has 66 assay records; the pinned repository README says 74, so this value is marked Conflicted.",
                "reporting_status": "reported"
              },
              {
                "basis": "proteins",
                "count": 2525,
                "exclusive": false,
                "exhaustive": false,
                "id": "clinical-substitution-proteins",
                "label": "Clinical substitution proteins",
                "notes": "The v1.3 clinical reference file has one row per protein.",
                "reporting_status": "reported"
              },
              {
                "basis": "proteins",
                "count": 1555,
                "exclusive": false,
                "exhaustive": false,
                "id": "clinical-indel-proteins",
                "label": "Clinical indel proteins",
                "notes": "The v1.3 clinical indel reference file has one row per protein.",
                "reporting_status": "reported"
              },
              {
                "basis": "assays",
                "count": 13,
                "exclusive": false,
                "exhaustive": false,
                "id": "dms-substitution-binding-assays",
                "label": "DMS substitution assays in the Binding function category",
                "notes": "The v1.3 reference file contains 13 rows with coarse_selection_type=Binding; target type is not split into protein-protein versus protein-ligand counts.",
                "reporting_status": "reported"
              }
            ],
            "total": null
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "Reference files, processed/raw variants, model scores, and MSAs are public.",
        "biosafety_notes": "Clinical labels require careful interpretation; the creator paper notes insufficient validation can limit medical use.",
        "grader": "Official deterministic scoring reports full-dataset AUROC and AUPRC because many proteins lack both classes.",
        "level": "fully-open",
        "license": "MIT",
        "tasks": "The processed clinical indel benchmark is publicly downloadable."
      },
      "aliases": [
        "ProteinGym clinical indel benchmark"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Protein counts were checked against the final paper and v1.3 reference file.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction",
        "classification"
      ],
      "coverage_notes": [],
      "domains": [
        "protein-sequence",
        "clinical-translational"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "proteingym-clinical-indel-evidence-identity",
          "locator": {
            "note": "Track definition, identity, domains, modalities, classification format, metric, and access.",
            "type": "section",
            "value": "pp. 1–10 and 31, Sections 3–5 and Appendix A.3.2"
          },
          "source_id": "proteingym-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/kind",
            "/summary",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "proteingym-clinical-indel-evidence-v10",
          "locator": {
            "note": "1,555 clinical indel proteins in v1.0.",
            "type": "table",
            "value": "pp. 6 and 29, Tables 1 and A1"
          },
          "source_id": "proteingym-paper",
          "source_type": "work",
          "supports": [
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "proteingym-clinical-indel-evidence-v13",
          "locator": {
            "note": "1,555 reference rows, one per clinical protein.",
            "type": "repository-path",
            "value": "clinical_indels.csv in Zenodo record 15293562"
          },
          "source_id": "proteingym-clinical-indel-v13-resource",
          "source_type": "resource",
          "supports": [
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "proteingym-clinical-indel-evidence-access",
          "locator": {
            "note": "Current version, downloads, scoring, implementation, and MIT license.",
            "type": "repository-path",
            "value": "README.md Overview, Results, Resources, Usage and reproducibility, and License"
          },
          "source_id": "proteingym-clinical-indel-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/resources",
            "/implementations"
          ]
        }
      ],
      "field_status": [],
      "id": "proteingym-clinical-indels",
      "implementations": [
        {
          "commit": "1f8de974dead8ff7501eff087b725d14a965e9f9",
          "framework": "ProteinGym clinical scoring pipeline",
          "notes": "Official clinical processing and evaluation notebooks.",
          "status": "official",
          "url": "https://github.com/OATML-Markslab/ProteinGym/tree/1f8de974dead8ff7501eff087b725d14a965e9f9/proteingym/clinical_benchmark_notebooks"
        }
      ],
      "kind": "track",
      "latest_version": "1.3",
      "modalities": [
        "protein-sequence",
        "table"
      ],
      "name": "ProteinGym Clinical Indels",
      "organizations": [
        "University of Oxford",
        "Harvard Medical School"
      ],
      "parent_id": "proteingym",
      "release_date": "2023-12-08",
      "resources": [
        {
          "access_notes": "Final creator paper for v1.0.",
          "id": "proteingym-clinical-indel-paper-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "snapshot",
            "url": "https://proceedings.neurips.cc/paper_files/paper/2023/file/cac723e5ff29f65e3fcbb0739ae91bee-Paper-Datasets_and_Benchmarks.pdf",
            "value": "sha256:a3b08cc4a6befd64620cf0f287d78d55a36dc2639955f52c5655f83833c50104"
          },
          "type": "paper",
          "url": "https://proceedings.neurips.cc/paper_files/paper/2023/file/cac723e5ff29f65e3fcbb0739ae91bee-Paper-Datasets_and_Benchmarks.pdf"
        },
        {
          "access_notes": "Official code and release index.",
          "id": "proteingym-clinical-indel-repository-resource",
          "last_checked": "2026-07-21",
          "license": "MIT",
          "pin": {
            "kind": "tag",
            "url": "https://github.com/OATML-Markslab/ProteinGym/tree/1f8de974dead8ff7501eff087b725d14a965e9f9",
            "value": "PG_v1.3"
          },
          "type": "repository",
          "url": "https://github.com/OATML-Markslab/ProteinGym"
        },
        {
          "access_notes": "Official v1.3 archive.",
          "id": "proteingym-clinical-indel-v13-resource",
          "last_checked": "2026-07-21",
          "license": "MIT",
          "pin": {
            "kind": "version",
            "url": "https://zenodo.org/records/15293562",
            "value": "1.3"
          },
          "type": "dataset",
          "url": "https://doi.org/10.5281/zenodo.15293562"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "1.3",
        "entries": [
          {
            "confidence": "high",
            "count": 1555,
            "count_basis": "Clinical proteins.",
            "count_ref": "/task_counts/total",
            "count_unit": "records",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "proteingym-clinical-indel-evidence-v13"
            ],
            "mapping_method": "official-track",
            "notes": "The registry count unit is proteins, not variants.",
            "reporting_status": "reported",
            "task_type_id": "protein-clinical-variant-interpretation"
          }
        ],
        "notes": "Formal clinical indel track.",
        "status": "complete"
      },
      "summary": "ProteinGym track for classifying short human clinical insertion and deletion variants against ClinVar and gnomAD-derived labels.",
      "task_counts": {
        "basis": "clinical proteins",
        "reporting_status": "reported",
        "subsets": [],
        "total": 1555
      },
      "task_formats": [
        "clinical variant classification"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Field-level track audit complete.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2023-12-08",
          "evidence_ids": [
            "proteingym-clinical-indel-evidence-v10"
          ],
          "formal_tracks": [],
          "id": "proteingym-clinical-indel-v10",
          "label": "1.0",
          "notes": "Creator-paper snapshot reporting approximately 3,000 short indel variants.",
          "release_date": "2023-12-08",
          "status": "superseded",
          "task_counts": {
            "basis": "clinical proteins",
            "reporting_status": "reported",
            "subsets": [],
            "total": 1555
          }
        },
        {
          "as_of": "2025-04-27",
          "evidence_ids": [
            "proteingym-clinical-indel-evidence-v13"
          ],
          "formal_tracks": [],
          "id": "proteingym-clinical-indel-v13",
          "label": "1.3",
          "notes": "Current reference file has one row per clinical protein.",
          "release_date": "2025-04-27",
          "status": "current",
          "task_counts": {
            "basis": "clinical proteins",
            "reporting_status": "reported",
            "subsets": [],
            "total": 1555
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "Reference files, processed/raw variants, model scores, and MSAs are public.",
        "biosafety_notes": "Clinical labels require careful interpretation; the creator paper notes insufficient validation can limit medical use.",
        "grader": "Official deterministic scoring reports gene-level AUROC and AUPRC analyses.",
        "level": "fully-open",
        "license": "MIT",
        "tasks": "The processed clinical substitution benchmark is publicly downloadable."
      },
      "aliases": [
        "ProteinGym clinical substitution benchmark"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Protein counts were checked against the final paper and v1.3 reference file.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction",
        "classification"
      ],
      "coverage_notes": [],
      "domains": [
        "protein-sequence",
        "clinical-translational"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "proteingym-clinical-sub-evidence-identity",
          "locator": {
            "note": "Track definition, identity, domains, modalities, classification format, metric, and access.",
            "type": "section",
            "value": "pp. 1–9 and 31, Sections 3–5 and Appendix A.3.2"
          },
          "source_id": "proteingym-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/kind",
            "/summary",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "proteingym-clinical-sub-evidence-v10",
          "locator": {
            "note": "2,525 clinical substitution proteins in v1.0.",
            "type": "table",
            "value": "pp. 6 and 29, Tables 1 and A1"
          },
          "source_id": "proteingym-paper",
          "source_type": "work",
          "supports": [
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "proteingym-clinical-sub-evidence-v13",
          "locator": {
            "note": "2,525 reference rows, one per clinical protein.",
            "type": "repository-path",
            "value": "clinical_substitutions.csv in Zenodo record 15293562"
          },
          "source_id": "proteingym-clinical-sub-v13-resource",
          "source_type": "resource",
          "supports": [
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "proteingym-clinical-sub-evidence-access",
          "locator": {
            "note": "Current version, downloads, scoring, implementation, and MIT license.",
            "type": "repository-path",
            "value": "README.md Overview, Results, Resources, Usage and reproducibility, and License"
          },
          "source_id": "proteingym-clinical-sub-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/resources",
            "/implementations"
          ]
        }
      ],
      "field_status": [],
      "id": "proteingym-clinical-substitutions",
      "implementations": [
        {
          "commit": "1f8de974dead8ff7501eff087b725d14a965e9f9",
          "framework": "ProteinGym clinical scoring pipeline",
          "notes": "Official clinical processing and evaluation notebooks.",
          "status": "official",
          "url": "https://github.com/OATML-Markslab/ProteinGym/tree/1f8de974dead8ff7501eff087b725d14a965e9f9/proteingym/clinical_benchmark_notebooks"
        }
      ],
      "kind": "track",
      "latest_version": "1.3",
      "modalities": [
        "protein-sequence",
        "table"
      ],
      "name": "ProteinGym Clinical Substitutions",
      "organizations": [
        "University of Oxford",
        "Harvard Medical School"
      ],
      "parent_id": "proteingym",
      "release_date": "2023-12-08",
      "resources": [
        {
          "access_notes": "Final creator paper for v1.0.",
          "id": "proteingym-clinical-sub-paper-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "snapshot",
            "url": "https://proceedings.neurips.cc/paper_files/paper/2023/file/cac723e5ff29f65e3fcbb0739ae91bee-Paper-Datasets_and_Benchmarks.pdf",
            "value": "sha256:a3b08cc4a6befd64620cf0f287d78d55a36dc2639955f52c5655f83833c50104"
          },
          "type": "paper",
          "url": "https://proceedings.neurips.cc/paper_files/paper/2023/file/cac723e5ff29f65e3fcbb0739ae91bee-Paper-Datasets_and_Benchmarks.pdf"
        },
        {
          "access_notes": "Official code and release index.",
          "id": "proteingym-clinical-sub-repository-resource",
          "last_checked": "2026-07-21",
          "license": "MIT",
          "pin": {
            "kind": "tag",
            "url": "https://github.com/OATML-Markslab/ProteinGym/tree/1f8de974dead8ff7501eff087b725d14a965e9f9",
            "value": "PG_v1.3"
          },
          "type": "repository",
          "url": "https://github.com/OATML-Markslab/ProteinGym"
        },
        {
          "access_notes": "Official v1.3 archive.",
          "id": "proteingym-clinical-sub-v13-resource",
          "last_checked": "2026-07-21",
          "license": "MIT",
          "pin": {
            "kind": "version",
            "url": "https://zenodo.org/records/15293562",
            "value": "1.3"
          },
          "type": "dataset",
          "url": "https://doi.org/10.5281/zenodo.15293562"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "1.3",
        "entries": [
          {
            "confidence": "high",
            "count": 2525,
            "count_basis": "Clinical proteins.",
            "count_ref": "/task_counts/total",
            "count_unit": "records",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "proteingym-clinical-sub-evidence-v13"
            ],
            "mapping_method": "official-track",
            "notes": "The registry count unit is proteins, not variants.",
            "reporting_status": "reported",
            "task_type_id": "protein-clinical-variant-interpretation"
          }
        ],
        "notes": "Formal clinical substitution track.",
        "status": "complete"
      },
      "summary": "ProteinGym track for classifying expert-annotated human clinical substitution variants on a per-protein basis.",
      "task_counts": {
        "basis": "clinical proteins",
        "reporting_status": "reported",
        "subsets": [],
        "total": 2525
      },
      "task_formats": [
        "clinical variant classification"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Field-level track audit complete.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2023-12-08",
          "evidence_ids": [
            "proteingym-clinical-sub-evidence-v10"
          ],
          "formal_tracks": [],
          "id": "proteingym-clinical-sub-v10",
          "label": "1.0",
          "notes": "Creator-paper snapshot reporting approximately 63,000 variants.",
          "release_date": "2023-12-08",
          "status": "superseded",
          "task_counts": {
            "basis": "clinical proteins",
            "reporting_status": "reported",
            "subsets": [],
            "total": 2525
          }
        },
        {
          "as_of": "2025-04-27",
          "evidence_ids": [
            "proteingym-clinical-sub-evidence-v13"
          ],
          "formal_tracks": [],
          "id": "proteingym-clinical-sub-v13",
          "label": "1.3",
          "notes": "Current reference file has one row per clinical protein.",
          "release_date": "2025-04-27",
          "status": "current",
          "task_counts": {
            "basis": "clinical proteins",
            "reporting_status": "reported",
            "subsets": [],
            "total": 2525
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "Reference files, raw/processed assays, model scores, MSAs, and CV folds are public.",
        "biosafety_notes": "The creator paper discusses dual-use risk for protein design and fitness models; this registry mirrors metadata and links only.",
        "grader": "Official deterministic scoring scripts compute zero-shot and supervised indel metrics.",
        "level": "fully-open",
        "license": "MIT",
        "tasks": "The versioned v1.3 archive exposes 66 processed indel assay files."
      },
      "aliases": [
        "ProteinGym indel DMS benchmark"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "The versioned release archive takes precedence, while the README disagreement remains visible.",
        "status": "audited-with-caveats",
        "unresolved_fields": 2
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction",
        "regression",
        "classification",
        "design",
        "optimization"
      ],
      "coverage_notes": [
        {
          "count": null,
          "coverage": "explicitly-in-scope",
          "notes": "The track uses fitness-prediction and design-oriented metrics, but design is not a separately counted subset.",
          "reporting_status": "not_reported",
          "tag": "protein-design"
        },
        {
          "count": 0,
          "coverage": "not-in-scope",
          "notes": "The official generic Binding function category contains zero indel assays in the paper and current reference file.",
          "reporting_status": "reported",
          "tag": "protein-protein-binding"
        },
        {
          "count": 0,
          "coverage": "not-in-scope",
          "notes": "The official generic Binding function category contains zero indel assays in the paper and current reference file.",
          "reporting_status": "reported",
          "tag": "protein-ligand-binding"
        }
      ],
      "domains": [
        "protein-sequence",
        "protein-design"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "proteingym-dms-indel-evidence-identity",
          "locator": {
            "note": "Track definition, parent relationship, domains, modalities, formats, capabilities, and access.",
            "type": "section",
            "value": "pp. 1–10, Figure 1 and Sections 3–5"
          },
          "source_id": "proteingym-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/kind",
            "/summary",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "proteingym-dms-indel-evidence-v10",
          "locator": {
            "note": "66 v1.0 indel assays and zero Binding indel assays.",
            "type": "table",
            "value": "pp. 6, 29–30; Tables 1, A1, and A2"
          },
          "source_id": "proteingym-paper",
          "source_type": "work",
          "supports": [
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "proteingym-dms-indel-evidence-v13",
          "locator": {
            "note": "Both expose 66 assay records; no coarse_selection_type=Binding row.",
            "type": "repository-path",
            "value": "DMS_indels.csv and DMS_ProteinGym_indels.zip in Zenodo record 15293562"
          },
          "source_id": "proteingym-dms-indel-v13-resource",
          "source_type": "resource",
          "supports": [
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets",
            "/scientific_task_classification/entries/0",
            "/scientific_task_classification/entries/1"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "proteingym-dms-indel-evidence-readme",
          "locator": {
            "note": "States 74 DMS indel assays.",
            "type": "repository-path",
            "value": "README.md Overview at tag PG_v1.3"
          },
          "source_id": "proteingym-dms-indel-repository-resource",
          "source_type": "resource",
          "supports": [
            "/task_counts/total",
            "/versions/1/task_counts/total"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "proteingym-dms-indel-evidence-access",
          "locator": {
            "note": "Current version, downloads, public scoring, implementation pin, and MIT license.",
            "type": "repository-path",
            "value": "README.md Resources, Results, Usage and reproducibility, Releases, and License"
          },
          "source_id": "proteingym-dms-indel-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/resources",
            "/implementations"
          ]
        }
      ],
      "field_status": [
        {
          "confidence": "medium",
          "evidence_ids": [
            "proteingym-dms-indel-evidence-v13",
            "proteingym-dms-indel-evidence-readme"
          ],
          "path": "/task_counts/total",
          "reason": "The v1.3 archive/reference file contains 66 assays while the official PG_v1.3 README states 74.",
          "status": "conflicted"
        },
        {
          "confidence": "medium",
          "evidence_ids": [
            "proteingym-dms-indel-evidence-v13",
            "proteingym-dms-indel-evidence-readme"
          ],
          "path": "/versions/1/task_counts/total",
          "reason": "The v1.3 archive/reference file contains 66 assays while the official PG_v1.3 README states 74.",
          "status": "conflicted"
        }
      ],
      "id": "proteingym-dms-indels",
      "implementations": [
        {
          "commit": "1f8de974dead8ff7501eff087b725d14a965e9f9",
          "framework": "ProteinGym DMS indel scoring pipeline",
          "notes": "Includes the official indel performance script.",
          "status": "official",
          "url": "https://github.com/OATML-Markslab/ProteinGym/tree/1f8de974dead8ff7501eff087b725d14a965e9f9/scripts/scoring_DMS_zero_shot"
        }
      ],
      "kind": "track",
      "latest_version": "1.3",
      "modalities": [
        "protein-sequence",
        "table"
      ],
      "name": "ProteinGym DMS Indels",
      "organizations": [
        "University of Oxford",
        "Harvard Medical School"
      ],
      "parent_id": "proteingym",
      "release_date": "2023-12-08",
      "resources": [
        {
          "access_notes": "Final creator paper for v1.0.",
          "id": "proteingym-dms-indel-paper-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "snapshot",
            "url": "https://proceedings.neurips.cc/paper_files/paper/2023/file/cac723e5ff29f65e3fcbb0739ae91bee-Paper-Datasets_and_Benchmarks.pdf",
            "value": "sha256:a3b08cc4a6befd64620cf0f287d78d55a36dc2639955f52c5655f83833c50104"
          },
          "type": "paper",
          "url": "https://proceedings.neurips.cc/paper_files/paper/2023/file/cac723e5ff29f65e3fcbb0739ae91bee-Paper-Datasets_and_Benchmarks.pdf"
        },
        {
          "access_notes": "Official scoring code, README, and release index.",
          "id": "proteingym-dms-indel-repository-resource",
          "last_checked": "2026-07-21",
          "license": "MIT",
          "pin": {
            "kind": "tag",
            "url": "https://github.com/OATML-Markslab/ProteinGym/tree/1f8de974dead8ff7501eff087b725d14a965e9f9",
            "value": "PG_v1.3"
          },
          "type": "repository",
          "url": "https://github.com/OATML-Markslab/ProteinGym"
        },
        {
          "access_notes": "Official v1.3 archive.",
          "id": "proteingym-dms-indel-v13-resource",
          "last_checked": "2026-07-21",
          "license": "MIT",
          "pin": {
            "kind": "version",
            "url": "https://zenodo.org/records/15293562",
            "value": "1.3"
          },
          "type": "dataset",
          "url": "https://doi.org/10.5281/zenodo.15293562"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "1.3",
        "entries": [
          {
            "confidence": "high",
            "count": 66,
            "count_basis": "DMS assays.",
            "count_ref": "/task_counts/total",
            "count_unit": "assays",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "proteingym-dms-indel-evidence-v13"
            ],
            "mapping_method": "official-track",
            "notes": "The underlying benchmark field carries the audit caveat.",
            "reporting_status": "reported",
            "task_type_id": "protein-mutation-effect-prediction"
          },
          {
            "confidence": "high",
            "count": 66,
            "count_basis": "DMS assays.",
            "count_ref": "/task_counts/total",
            "count_unit": "assays",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "proteingym-dms-indel-evidence-v13"
            ],
            "mapping_method": "official-track",
            "notes": "Overlapping claim; not an additive task total.",
            "reporting_status": "reported",
            "task_type_id": "protein-fitness-prediction"
          }
        ],
        "notes": "Formal DMS indel track; source conflicts on assay count remain flagged on the benchmark.",
        "status": "complete"
      },
      "summary": "ProteinGym track for predicting experimental fitness measurements of insertion and deletion mutants across deep-mutational-scanning assays.",
      "task_counts": {
        "basis": "DMS assays",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "assays",
            "count": 0,
            "exclusive": false,
            "exhaustive": false,
            "id": "binding-function-assays",
            "label": "Binding function category",
            "notes": "Neither Table A2 nor the v1.3 coarse_selection_type field contains an indel Binding assay.",
            "reporting_status": "reported"
          }
        ],
        "total": 66
      },
      "task_formats": [
        "zero-shot mutation-effect prediction",
        "supervised mutation-effect prediction"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Field-level track audit complete with an unresolved official-source count conflict.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2023-12-08",
          "evidence_ids": [
            "proteingym-dms-indel-evidence-v10"
          ],
          "formal_tracks": [],
          "id": "proteingym-dms-indel-v10",
          "label": "1.0",
          "notes": "Creator-paper snapshot with 289,000 indel mutants.",
          "release_date": "2023-12-08",
          "status": "superseded",
          "task_counts": {
            "basis": "DMS assays",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "assays",
                "count": 0,
                "exclusive": false,
                "exhaustive": false,
                "id": "binding-function-assays",
                "label": "Binding function category",
                "notes": "Table A2 value.",
                "reporting_status": "reported"
              }
            ],
            "total": 66
          }
        },
        {
          "as_of": "2025-04-27",
          "evidence_ids": [
            "proteingym-dms-indel-evidence-v13",
            "proteingym-dms-indel-evidence-readme"
          ],
          "formal_tracks": [],
          "id": "proteingym-dms-indel-v13",
          "label": "1.3",
          "notes": "The archive has 66 assays; the pinned README says 74 and is preserved as a conflict.",
          "release_date": "2025-04-27",
          "status": "current",
          "task_counts": {
            "basis": "DMS assays",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "assays",
                "count": 0,
                "exclusive": false,
                "exhaustive": false,
                "id": "binding-function-assays",
                "label": "Binding function category",
                "notes": "No v1.3 coarse_selection_type=Binding row.",
                "reporting_status": "reported"
              }
            ],
            "total": 66
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "Reference files, raw/processed assays, model scores, MSAs, structures, and CV folds are public.",
        "biosafety_notes": "The creator paper discusses dual-use risk for protein design and fitness models; this registry mirrors metadata and links only.",
        "grader": "Official deterministic scoring scripts compute zero-shot and supervised metrics.",
        "level": "fully-open",
        "license": "MIT",
        "tasks": "All 217 processed v1.3 assay tables are publicly downloadable."
      },
      "aliases": [
        "ProteinGym substitution DMS benchmark"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Assay counts and the versioned Binding-category change were checked against the final paper and v1.3 reference file.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction",
        "regression",
        "classification",
        "design",
        "optimization"
      ],
      "coverage_notes": [
        {
          "count": null,
          "coverage": "explicitly-in-scope",
          "notes": "NDCG@10% and top-10% recall are the design-oriented metrics; design is not a separately counted assay subset.",
          "reporting_status": "not_reported",
          "tag": "protein-design"
        },
        {
          "count": null,
          "coverage": "explicitly-in-scope",
          "notes": "The generic Binding function category has 13 current assays, but target type is not split, so no ligand-only count is inferred.",
          "reporting_status": "not_reported",
          "tag": "protein-ligand-binding"
        },
        {
          "count": null,
          "coverage": "observed",
          "notes": "Some source assays concern protein-protein binding, but the official category is generic Binding and no protein-protein-only count is published.",
          "reporting_status": "not_reported",
          "tag": "protein-protein-binding"
        }
      ],
      "domains": [
        "protein-sequence",
        "protein-design",
        "protein-ligand-binding"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "proteingym-paper"
      ],
      "evaluation_run_ids": [
        "proteingym-v10-dms-substitutions-zero-shot"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "proteingym-dms-sub-evidence-identity",
          "locator": {
            "note": "Track definition, parent relationship, domains, modalities, formats, capabilities, metrics, and access.",
            "type": "section",
            "value": "pp. 1–9, Figure 1 and Sections 3–5"
          },
          "source_id": "proteingym-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/kind",
            "/summary",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "proteingym-dms-sub-evidence-v10",
          "locator": {
            "note": "217 substitution assays and 14 Binding assays in v1.0.",
            "type": "table",
            "value": "pp. 6, 29–30; Tables 1, A1, and A2"
          },
          "source_id": "proteingym-paper",
          "source_type": "work",
          "supports": [
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "proteingym-dms-sub-evidence-v13",
          "locator": {
            "note": "217 rows total and 13 rows with coarse_selection_type=Binding.",
            "type": "repository-path",
            "value": "DMS_substitutions.csv in Zenodo record 15293562"
          },
          "source_id": "proteingym-dms-sub-v13-resource",
          "source_type": "resource",
          "supports": [
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets",
            "/scientific_task_classification/entries/0",
            "/scientific_task_classification/entries/1"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "proteingym-dms-sub-evidence-access",
          "locator": {
            "note": "Current version, downloads, public scoring, implementation pin, and MIT license.",
            "type": "repository-path",
            "value": "README.md Resources, Results, Usage and reproducibility, Releases, and License"
          },
          "source_id": "proteingym-dms-sub-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/resources",
            "/implementations"
          ]
        }
      ],
      "field_status": [],
      "id": "proteingym-dms-substitutions",
      "implementations": [
        {
          "commit": "1f8de974dead8ff7501eff087b725d14a965e9f9",
          "framework": "ProteinGym DMS scoring pipeline",
          "notes": "Separate scripts cover substitution zero-shot and supervised scoring.",
          "status": "official",
          "url": "https://github.com/OATML-Markslab/ProteinGym/tree/1f8de974dead8ff7501eff087b725d14a965e9f9/scripts/scoring_DMS_zero_shot"
        }
      ],
      "kind": "track",
      "latest_version": "1.3",
      "modalities": [
        "protein-sequence",
        "structure-3d",
        "table"
      ],
      "name": "ProteinGym DMS Substitutions",
      "organizations": [
        "University of Oxford",
        "Harvard Medical School"
      ],
      "parent_id": "proteingym",
      "release_date": "2023-12-08",
      "resources": [
        {
          "access_notes": "Final creator paper for v1.0.",
          "id": "proteingym-dms-sub-paper-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "snapshot",
            "url": "https://proceedings.neurips.cc/paper_files/paper/2023/file/cac723e5ff29f65e3fcbb0739ae91bee-Paper-Datasets_and_Benchmarks.pdf",
            "value": "sha256:a3b08cc4a6befd64620cf0f287d78d55a36dc2639955f52c5655f83833c50104"
          },
          "type": "paper",
          "url": "https://proceedings.neurips.cc/paper_files/paper/2023/file/cac723e5ff29f65e3fcbb0739ae91bee-Paper-Datasets_and_Benchmarks.pdf"
        },
        {
          "access_notes": "Official scoring code and release index.",
          "id": "proteingym-dms-sub-repository-resource",
          "last_checked": "2026-07-21",
          "license": "MIT",
          "pin": {
            "kind": "tag",
            "url": "https://github.com/OATML-Markslab/ProteinGym/tree/1f8de974dead8ff7501eff087b725d14a965e9f9",
            "value": "PG_v1.3"
          },
          "type": "repository",
          "url": "https://github.com/OATML-Markslab/ProteinGym"
        },
        {
          "access_notes": "Official v1.3 archive.",
          "id": "proteingym-dms-sub-v13-resource",
          "last_checked": "2026-07-21",
          "license": "MIT",
          "pin": {
            "kind": "version",
            "url": "https://zenodo.org/records/15293562",
            "value": "1.3"
          },
          "type": "dataset",
          "url": "https://doi.org/10.5281/zenodo.15293562"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "1.3",
        "entries": [
          {
            "confidence": "high",
            "count": 217,
            "count_basis": "DMS assays.",
            "count_ref": "/task_counts/total",
            "count_unit": "assays",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "proteingym-dms-sub-evidence-v13"
            ],
            "mapping_method": "official-track",
            "notes": "Assay count; not mutant-record count.",
            "reporting_status": "reported",
            "task_type_id": "protein-mutation-effect-prediction"
          },
          {
            "confidence": "high",
            "count": 217,
            "count_basis": "DMS assays.",
            "count_ref": "/task_counts/total",
            "count_unit": "assays",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "proteingym-dms-sub-evidence-v13"
            ],
            "mapping_method": "official-track",
            "notes": "Overlapping scientific-task claim; never summed with mutation-effect coverage.",
            "reporting_status": "reported",
            "task_type_id": "protein-fitness-prediction"
          }
        ],
        "notes": "Formal DMS substitution track.",
        "status": "complete"
      },
      "summary": "ProteinGym track for predicting experimental fitness measurements of substitution mutants across deep-mutational-scanning assays.",
      "task_counts": {
        "basis": "DMS assays",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "assays",
            "count": 13,
            "exclusive": false,
            "exhaustive": false,
            "id": "binding-function-assays",
            "label": "Binding function category",
            "notes": "v1.3 coarse_selection_type=Binding; v1.0 reported 14.",
            "reporting_status": "reported"
          }
        ],
        "total": 217
      },
      "task_formats": [
        "zero-shot mutation-effect prediction",
        "supervised mutation-effect prediction"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Field-level track audit complete.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2023-12-08",
          "evidence_ids": [
            "proteingym-dms-sub-evidence-v10"
          ],
          "formal_tracks": [],
          "id": "proteingym-dms-sub-v10",
          "label": "1.0",
          "notes": "Creator-paper snapshot with 2.4 million substitution mutants.",
          "release_date": "2023-12-08",
          "status": "superseded",
          "task_counts": {
            "basis": "DMS assays",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "assays",
                "count": 14,
                "exclusive": false,
                "exhaustive": false,
                "id": "binding-function-assays",
                "label": "Binding function category",
                "notes": "Table A2 value.",
                "reporting_status": "reported"
              }
            ],
            "total": 217
          }
        },
        {
          "as_of": "2025-04-27",
          "evidence_ids": [
            "proteingym-dms-sub-evidence-v13"
          ],
          "formal_tracks": [],
          "id": "proteingym-dms-sub-v13",
          "label": "1.3",
          "notes": "Current official dataset snapshot; intermediate release history remains documented on the parent suite.",
          "release_date": "2025-04-27",
          "status": "current",
          "task_counts": {
            "basis": "DMS assays",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "assays",
                "count": 13,
                "exclusive": false,
                "exhaustive": false,
                "id": "binding-function-assays",
                "label": "Binding function category",
                "notes": "coarse_selection_type=Binding in the versioned reference file.",
                "reporting_status": "reported"
              }
            ],
            "total": 217
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "Each record contains question, options, correct-answer label, and explanation; the current snapshot has no topical-category or raw-sequence field.",
        "biosafety_notes": "Questions cover protein science and engineering literature. The registry mirrors no questions, answer explanations, sequences, training corpora, or model outputs.",
        "grader": "The official runner extracts the first integer from a response and compares it exactly with the answer-option integer.",
        "level": "fully-open",
        "license": "Apache-2.0 on the current Hugging Face dataset card and official code repository; the paper dataset card separately says Toursun Synbio metadata are CC BY 4.0.",
        "tasks": "The complete evaluation JSON and CSV are publicly downloadable without gating from the creator Hugging Face repository."
      },
      "aliases": [
        "Protein LM Bench",
        "ProteinLMBenchmark"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "The versioned dataset takes precedence for the current record. Official-source conflicts over all-six-choice format and license remain machine-readable; topical coverage cannot be counted because the release has no category labels.",
        "status": "audited-with-caveats",
        "unresolved_fields": 4
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "knowledge",
        "scientific-reasoning",
        "prediction"
      ],
      "coverage_notes": [
        {
          "count": null,
          "coverage": "explicitly-in-scope",
          "notes": "The paper dataset card names Protein design as a possible benchmark task, but the released evaluation file has no topical category field, so no design-question count is inferred.",
          "reporting_status": "not_reported",
          "tag": "protein-design"
        },
        {
          "count": null,
          "coverage": "explicitly-in-scope",
          "notes": "The paper claims sequence-understanding coverage, but the current evaluation JSON exposes only question, options, answer, and explanation text fields and contains no explicit raw-sequence input field.",
          "reporting_status": "not_reported",
          "tag": "protein-sequence"
        },
        {
          "count": null,
          "coverage": "observed",
          "notes": "Official questions discuss protein-protein interactions and antibody binding, but no official topical labels support a standalone count.",
          "reporting_status": "not_reported",
          "tag": "protein-protein-binding"
        },
        {
          "count": null,
          "coverage": "observed",
          "notes": "Official questions discuss ligand binding, but no official topical labels support a standalone count.",
          "reporting_status": "not_reported",
          "tag": "protein-ligand-binding"
        }
      ],
      "domains": [
        "protein-science",
        "protein-sequence",
        "protein-structure",
        "protein-design",
        "protein-protein-binding",
        "protein-ligand-binding"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "proteinlmbench-paper"
      ],
      "evaluation_run_ids": [
        "proteinlmbench-creator-full"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "proteinlmbench-evidence-paper",
          "locator": {
            "note": "944 count, claimed six-choice format, topic scope, construction/verification, evaluated models/results, no repeats/seeds, license statement, and paper-version snapshot.",
            "type": "section",
            "value": "Abstract; Sections 3.2, 4.3, 5-6; Table 3; Appendix B.5-B.7 and C"
          },
          "source_id": "proteinlmbench-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/organizations",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets/4/count",
            "/coverage_notes",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/versions/1/task_counts/subsets/4/count",
            "/versions/2/release_date",
            "/versions/2/task_counts/total",
            "/versions/2/task_counts/basis",
            "/versions/2/task_counts/subsets",
            "/scientific_task_classification/entries/0",
            "/scientific_task_classification/entries/1",
            "/scientific_task_classification/entries/2"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "proteinlmbench-evidence-dataset-history",
          "locator": {
            "note": "Initial CSV release date/counts and later JSON restoration.",
            "type": "release",
            "value": "Hugging Face commits c59f90c91e215bae673574c57c9a6c4a9f6aa87b and f1397963c7f727a4a2f00cdd691e6e219c36e992"
          },
          "source_id": "proteinlmbench-dataset-resource",
          "source_type": "resource",
          "supports": [
            "/release_date",
            "/versions/0/release_date",
            "/versions/0/as_of",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "proteinlmbench-evidence-current-dataset",
          "locator": {
            "note": "944 records; fields question/options/answer/explanation; option counts 2:3, 3:21, 4:42, 5:1, 6:871, 7:2, 8:1, 10:3; Apache-2.0 card.",
            "type": "repository-path",
            "value": "ProteinLMBench.json and README.md at f1397963c7f727a4a2f00cdd691e6e219c36e992"
          },
          "source_id": "proteinlmbench-dataset-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/task_counts/subsets/4/count",
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/license",
            "/resources",
            "/versions/1/release_date",
            "/versions/1/as_of",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets",
            "/versions/1/task_counts/subsets/4/count"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "proteinlmbench-evidence-runner",
          "locator": {
            "note": "Manual expert review statement, public prompt-only runner, parser/grader, and Apache-2.0 license.",
            "type": "repository-path",
            "value": "README.md; LICENSE; benchmark/benchmark_your_model.py at d8586e22ff85f6805edea0bbc23002aaccf525c4"
          },
          "source_id": "proteinlmbench-repository-resource",
          "source_type": "resource",
          "supports": [
            "/access/tasks",
            "/access/grader",
            "/access/license",
            "/resources",
            "/implementations"
          ]
        }
      ],
      "field_status": [
        {
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-evidence-paper",
            "proteinlmbench-evidence-current-dataset"
          ],
          "path": "/task_formats",
          "reason": "The paper calls every item six-choice, while the current official JSON contains 2-10 options per record.",
          "status": "conflicted"
        },
        {
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-evidence-paper",
            "proteinlmbench-evidence-current-dataset"
          ],
          "path": "/task_counts/subsets/4/count",
          "reason": "Current JSON has 871 six-choice records; paper v2 claims all 944 are six-choice.",
          "status": "conflicted"
        },
        {
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-evidence-paper",
            "proteinlmbench-evidence-current-dataset"
          ],
          "path": "/versions/1/task_counts/subsets/4/count",
          "reason": "Current JSON has 871 six-choice records; paper v2 claims all 944 are six-choice.",
          "status": "conflicted"
        },
        {
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-evidence-paper",
            "proteinlmbench-evidence-current-dataset"
          ],
          "path": "/access/license",
          "reason": "Current official dataset card and repository say Apache-2.0; the paper dataset card says Toursun Synbio metadata are CC BY 4.0. Both statements are retained.",
          "status": "conflicted"
        }
      ],
      "id": "proteinlmbench",
      "implementations": [
        {
          "commit": "d8586e22ff85f6805edea0bbc23002aaccf525c4",
          "framework": "ProteinLMBenchmark official runner",
          "notes": "Public prompt template, temperature 0.1, 20-new-token limit, first-integer answer parser, and exact accuracy scorer.",
          "status": "official",
          "url": "https://github.com/tsynbio/ProteinLMDataset/blob/d8586e22ff85f6805edea0bbc23002aaccf525c4/benchmark/benchmark_your_model.py"
        }
      ],
      "kind": "dataset",
      "latest_version": "hf-f139796",
      "modalities": [
        "text"
      ],
      "name": "ProteinLMBench",
      "organizations": [
        "Toursun Synbio",
        "Johns Hopkins University",
        "University of Cambridge",
        "Shanghai Institute for Biomedical and Pharmaceutical Technologies",
        "Shanghai AI Laboratory",
        "Shanghai Jiao Tong University",
        "UNSW Sydney"
      ],
      "parent_id": null,
      "release_date": "2024-04-29",
      "resources": [
        {
          "access_notes": "Creator preprint v2 with benchmark construction, evaluation results, dataset card, and training details.",
          "id": "proteinlmbench-paper-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "snapshot",
            "url": "https://arxiv.org/pdf/2406.05540",
            "value": "sha256:161a6e71e4e3321c91bd884ab2c906704c9110dbd5d3672651689e9369ca98dd"
          },
          "type": "paper",
          "url": "https://arxiv.org/pdf/2406.05540"
        },
        {
          "access_notes": "Official versioned dataset. ProteinLMBench.json is 1,087,875 bytes with SHA256 f2dce5c1b54e8ce73897a4af48ae8ba05507076d511b9d02d57cddf31c42ed96 at the pin.",
          "id": "proteinlmbench-dataset-resource",
          "last_checked": "2026-07-21",
          "license": "Apache-2.0",
          "pin": {
            "kind": "commit",
            "url": "https://huggingface.co/datasets/tsynbio/ProteinLMBench/tree/f1397963c7f727a4a2f00cdd691e6e219c36e992",
            "value": "f1397963c7f727a4a2f00cdd691e6e219c36e992"
          },
          "type": "dataset",
          "url": "https://huggingface.co/datasets/tsynbio/ProteinLMBench"
        },
        {
          "access_notes": "Official creator code for benchmark generation and prompt-only model evaluation.",
          "id": "proteinlmbench-repository-resource",
          "last_checked": "2026-07-21",
          "license": "Apache-2.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/tsynbio/ProteinLMDataset/tree/d8586e22ff85f6805edea0bbc23002aaccf525c4",
            "value": "d8586e22ff85f6805edea0bbc23002aaccf525c4"
          },
          "type": "repository",
          "url": "https://github.com/tsynbio/ProteinLMDataset"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "hf-f139796",
        "entries": [
          {
            "confidence": "high",
            "count": null,
            "count_basis": "Released ProteinLMBench question records.",
            "count_ref": null,
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "proteinlmbench-evidence-paper"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "Broad structure coverage only; folding is not inferred.",
            "reporting_status": "not_reported",
            "task_type_id": "protein-structure"
          },
          {
            "confidence": "high",
            "count": null,
            "count_basis": "Released ProteinLMBench question records.",
            "count_ref": null,
            "count_unit": "questions",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "proteinlmbench-evidence-paper"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "Broad protein-design topic; no sequence-generation count is available.",
            "reporting_status": "not_reported",
            "task_type_id": "protein-design"
          },
          {
            "confidence": "high",
            "count": null,
            "count_basis": "Released ProteinLMBench question records.",
            "count_ref": null,
            "count_unit": "questions",
            "coverage": "observed",
            "evidence_ids": [
              "proteinlmbench-evidence-paper"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "Official questions discuss multiple binding contexts without a topic field or binding-subtype counts.",
            "reporting_status": "not_reported",
            "task_type_id": "molecular-interaction-analysis"
          }
        ],
        "notes": "Official materials claim broad topical coverage, but the released question file has no official topic labels; no topical counts are inferred.",
        "status": "partial"
      },
      "summary": "A creator-curated set of 944 protein-science multiple-choice questions with answer explanations, generated from research literature and released for evaluating text LLM protein understanding.",
      "task_counts": {
        "basis": "question records in ProteinLMBench.json",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "questions grouped by options-array length",
            "count": 3,
            "exclusive": true,
            "exhaustive": true,
            "id": "proteinlmbench-two-choice",
            "label": "Two-choice questions",
            "notes": null,
            "reporting_status": "reported"
          },
          {
            "basis": "questions grouped by options-array length",
            "count": 21,
            "exclusive": true,
            "exhaustive": true,
            "id": "proteinlmbench-three-choice",
            "label": "Three-choice questions",
            "notes": null,
            "reporting_status": "reported"
          },
          {
            "basis": "questions grouped by options-array length",
            "count": 42,
            "exclusive": true,
            "exhaustive": true,
            "id": "proteinlmbench-four-choice",
            "label": "Four-choice questions",
            "notes": null,
            "reporting_status": "reported"
          },
          {
            "basis": "questions grouped by options-array length",
            "count": 1,
            "exclusive": true,
            "exhaustive": true,
            "id": "proteinlmbench-five-choice",
            "label": "Five-choice questions",
            "notes": null,
            "reporting_status": "reported"
          },
          {
            "basis": "questions grouped by options-array length",
            "count": 871,
            "exclusive": true,
            "exhaustive": true,
            "id": "proteinlmbench-six-choice",
            "label": "Six-choice questions",
            "notes": "The paper describes all 944 questions as six-choice; the current versioned JSON contains 871 six-choice records.",
            "reporting_status": "reported"
          },
          {
            "basis": "questions grouped by options-array length",
            "count": 2,
            "exclusive": true,
            "exhaustive": true,
            "id": "proteinlmbench-seven-choice",
            "label": "Seven-choice questions",
            "notes": "Both have answer option 7.",
            "reporting_status": "reported"
          },
          {
            "basis": "questions grouped by options-array length",
            "count": 1,
            "exclusive": true,
            "exhaustive": true,
            "id": "proteinlmbench-eight-choice",
            "label": "Eight-choice questions",
            "notes": null,
            "reporting_status": "reported"
          },
          {
            "basis": "questions grouped by options-array length",
            "count": 3,
            "exclusive": true,
            "exhaustive": true,
            "id": "proteinlmbench-ten-choice",
            "label": "Ten-choice questions",
            "notes": null,
            "reporting_status": "reported"
          }
        ],
        "total": 944
      },
      "task_formats": [
        "variable-choice multiple choice with 2 to 10 options in the current snapshot"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Audited against arXiv v2, the complete commit-pinned evaluation JSON and repository history, and the official evaluation runner.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2024-04-29",
          "evidence_ids": [
            "proteinlmbench-evidence-dataset-history"
          ],
          "formal_tracks": [],
          "id": "proteinlmbench-hf-c59f90c",
          "label": "hf-c59f90c",
          "notes": "Initial public benchmark-file snapshot; the repository itself was initialized earlier the same day.",
          "release_date": "2024-04-29",
          "status": "superseded",
          "task_counts": {
            "basis": "question rows in the initial public CSV",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "non-empty option columns in CSV",
                "count": 3,
                "exclusive": true,
                "exhaustive": true,
                "id": "proteinlmbench-two-choice",
                "label": "Two-choice questions",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "non-empty option columns in CSV",
                "count": 21,
                "exclusive": true,
                "exhaustive": true,
                "id": "proteinlmbench-three-choice",
                "label": "Three-choice questions",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "non-empty option columns in CSV",
                "count": 42,
                "exclusive": true,
                "exhaustive": true,
                "id": "proteinlmbench-four-choice",
                "label": "Four-choice questions",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "non-empty option columns in CSV",
                "count": 1,
                "exclusive": true,
                "exhaustive": true,
                "id": "proteinlmbench-five-choice",
                "label": "Five-choice questions",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "non-empty option columns in the six-option CSV schema",
                "count": 877,
                "exclusive": true,
                "exhaustive": true,
                "id": "proteinlmbench-six-choice",
                "label": "Six-choice questions visible in CSV",
                "notes": "The initial CSV schema had only six option columns; later JSON preserves 7, 8, and 10-option records.",
                "reporting_status": "reported"
              }
            ],
            "total": 944
          }
        },
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "proteinlmbench-evidence-current-dataset"
          ],
          "formal_tracks": [],
          "id": "proteinlmbench-hf-f139796",
          "label": "hf-f139796",
          "notes": "Current official revision. The evaluation JSON has 944 records and variable option counts; it contains no topical-category field.",
          "release_date": "2024-05-23",
          "status": "current",
          "task_counts": {
            "basis": "question records in ProteinLMBench.json",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions grouped by options-array length",
                "count": 3,
                "exclusive": true,
                "exhaustive": true,
                "id": "proteinlmbench-two-choice",
                "label": "Two-choice questions",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions grouped by options-array length",
                "count": 21,
                "exclusive": true,
                "exhaustive": true,
                "id": "proteinlmbench-three-choice",
                "label": "Three-choice questions",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions grouped by options-array length",
                "count": 42,
                "exclusive": true,
                "exhaustive": true,
                "id": "proteinlmbench-four-choice",
                "label": "Four-choice questions",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions grouped by options-array length",
                "count": 1,
                "exclusive": true,
                "exhaustive": true,
                "id": "proteinlmbench-five-choice",
                "label": "Five-choice questions",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions grouped by options-array length",
                "count": 871,
                "exclusive": true,
                "exhaustive": true,
                "id": "proteinlmbench-six-choice",
                "label": "Six-choice questions",
                "notes": "The paper describes all 944 questions as six-choice; the current versioned JSON contains 871 six-choice records.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions grouped by options-array length",
                "count": 2,
                "exclusive": true,
                "exhaustive": true,
                "id": "proteinlmbench-seven-choice",
                "label": "Seven-choice questions",
                "notes": "Both have answer option 7.",
                "reporting_status": "reported"
              },
              {
                "basis": "questions grouped by options-array length",
                "count": 1,
                "exclusive": true,
                "exhaustive": true,
                "id": "proteinlmbench-eight-choice",
                "label": "Eight-choice questions",
                "notes": null,
                "reporting_status": "reported"
              },
              {
                "basis": "questions grouped by options-array length",
                "count": 3,
                "exclusive": true,
                "exhaustive": true,
                "id": "proteinlmbench-ten-choice",
                "label": "Ten-choice questions",
                "notes": null,
                "reporting_status": "reported"
              }
            ],
            "total": 944
          }
        },
        {
          "as_of": "2024-07-08",
          "evidence_ids": [
            "proteinlmbench-evidence-paper"
          ],
          "formal_tracks": [],
          "id": "proteinlmbench-paper-v2",
          "label": "paper-v2",
          "notes": "Paper-defined evaluation snapshot used for Table 3; the manuscript does not pin an exact Hugging Face revision.",
          "release_date": "2024-07-08",
          "status": "active",
          "task_counts": {
            "basis": "questions described in creator preprint v2",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "questions",
                "count": 944,
                "exclusive": true,
                "exhaustive": true,
                "id": "proteinlmbench-paper-six-choice",
                "label": "Six-choice questions claimed by paper",
                "notes": "Conflicts with both the initial CSV and current JSON option-count distributions.",
                "reporting_status": "reported"
              }
            ],
            "total": 944
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "See the linked official creator resources.",
        "biosafety_notes": null,
        "grader": "Not established by the independently verified metadata claims.",
        "level": "fully-open",
        "license": null,
        "tasks": "See the linked official creator resources."
      },
      "aliases": [],
      "audit": {
        "audited_date": "2026-07-31",
        "notes": "Automated double-pass extraction plus deterministic official-resource pin; the benchmark license remains unverified and is published as null.",
        "status": "audited-with-caveats",
        "unresolved_fields": 2
      },
      "benchmark_use_ids": [
        "an-integrative-approach-to-protein-sequence-design-thr-rfah-benchmark-1-use",
        "an-integrative-approach-to-protein-sequence-design-thr-rfah-benchmark-2-use"
      ],
      "capabilities": [
        "design",
        "optimization"
      ],
      "coverage_notes": [],
      "domains": [
        "life-science",
        "protein-science",
        "protein-sequence",
        "protein-structure",
        "protein-design"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "an-integrative-approach-to-protein-sequence-design-thr"
      ],
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-31",
          "id": "rfah-benchmark-automated-metadata-1-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "cbdee6951e5532b57429398b51436fe8e2abfc9f78df018593e82380271ae3fc",
            "type": "section",
            "value": "Data Availability"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/access/level"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "rfah-benchmark-automated-metadata-2-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "3e19ed1d066f54d5605a7f436a578ce7bae045d00a5af665eb727e332bf86fc0",
            "type": "section",
            "value": "Results — The random resetting mutation operator results in slow convergence"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/capabilities",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "rfah-benchmark-automated-metadata-3-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "8d3f3b886ded53947f64600769dd861932936d0c320365566813b5c0d43e85a9",
            "type": "section",
            "value": "Results — The random resetting mutation operator results in slow convergence"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/domains"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "rfah-benchmark-automated-metadata-4-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "6e6903443229eedd52871ff2829f1ae65202e38c28b6be6dcf62d0c145ae0384",
            "type": "section",
            "value": "Data Availability"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/kind"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "rfah-benchmark-automated-metadata-5-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "f3f594888bef255b55222b36d3e680d7c43505d90a10c107aad536905e56e055",
            "type": "figure",
            "value": "Fig. 1"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/modalities"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "rfah-benchmark-automated-metadata-6-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "e5d671f353ebc91ccf2a583d7260d0931031355447522158d2919359486c39d6",
            "type": "section",
            "value": "Methods — Hypervolume"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/name"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "rfah-benchmark-automated-metadata-7-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "06c4d9bdb31c2ac3c4e3ceb12dff9a706e3bc8b0d242cbef9e52e7f81cb2a18e",
            "type": "other",
            "value": "Front matter — author-affiliation mapping"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/organizations"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "rfah-benchmark-automated-metadata-8-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "229f55e6020d2c34a78b2acb9235ab508a7af8bd056b5f980797f1bd9c59034a",
            "type": "figure",
            "value": "Fig. 1"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/summary"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "rfah-benchmark-automated-bibliographic-evidence",
          "locator": {
            "document_page": null,
            "note": "Resolved from the canonical paper identifier during intake.",
            "printed_page": null,
            "source_fragment_sha256": "5b45963847a151050eb2419c23684ca38e9325e2457145144b7ec0d2ea01d410",
            "type": "other",
            "value": "Crossref bibliographic metadata"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/release_date"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "rfah-benchmark-automated-count-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "7705b1f875bf17c0e6ab97786fb20f88491ceb770418c5f806fef0bfb856f59b",
            "type": "figure",
            "value": "Fig. 1"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/task_counts/total",
            "/task_counts/basis",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "rfah-benchmark-automated-subset-coverage-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "7705b1f875bf17c0e6ab97786fb20f88491ceb770418c5f806fef0bfb856f59b",
            "type": "figure",
            "value": "Fig. 1"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/task_counts/subsets",
            "/versions/0/task_counts/subsets"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "rfah-benchmark-automated-resource-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "cbdee6951e5532b57429398b51436fe8e2abfc9f78df018593e82380271ae3fc",
            "type": "section",
            "value": "Data Availability"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/resources",
            "/implementations",
            "/access/license"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "rfah-benchmark-automated-creator-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "6115205069f1a4999f060235c61c133f276780408241486c68044ff841a14049",
            "type": "other",
            "value": "Front matter — DOI"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/resources/0"
          ]
        },
        {
          "accessed_date": "2026-07-31",
          "id": "rfah-benchmark-automated-version-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "6115205069f1a4999f060235c61c133f276780408241486c68044ff841a14049",
            "type": "other",
            "value": "Front matter — DOI"
          },
          "source_id": "an-integrative-approach-to-protein-sequence-design-thr",
          "source_type": "work",
          "supports": [
            "/latest_version",
            "/versions/0"
          ]
        }
      ],
      "field_status": [
        {
          "confidence": "high",
          "evidence_ids": [
            "rfah-benchmark-automated-subset-coverage-evidence"
          ],
          "path": "/task_counts/subsets",
          "reason": "The double-pass review established the root item total but did not establish an exhaustive formal-subset inventory; the empty list must not be interpreted as evidence that the benchmark has no subsets.",
          "status": "provisional"
        },
        {
          "confidence": "high",
          "evidence_ids": [
            "rfah-benchmark-automated-resource-evidence"
          ],
          "path": "/access/license",
          "reason": "The double-pass review verified the official resource identity but did not establish a redistributable benchmark license; the value remains null pending source-level license verification.",
          "status": "provisional"
        }
      ],
      "id": "rfah-benchmark",
      "implementations": [
        {
          "commit": "b16a0ef3d5c44f65714b1a6c51826f6b4bdaa998",
          "framework": "official repository",
          "notes": "Commit resolved deterministically during paper intake.",
          "status": "official",
          "url": "https://github.com/luhong88/int_seq_des"
        }
      ],
      "kind": "dataset",
      "latest_version": "initial-release",
      "modalities": [
        "protein-sequence",
        "structure-3d"
      ],
      "name": "RfaH benchmark",
      "organizations": [
        "University of California, San Francisco",
        "Quantitative Biosciences Institute",
        "Chan Zuckerberg Biohub"
      ],
      "parent_id": null,
      "release_date": "2024-07-11",
      "resources": [
        {
          "access_notes": "Versioned creator source; full text is not mirrored.",
          "id": "rfah-benchmark-creator-paper-resource",
          "last_checked": "2026-07-31",
          "license": null,
          "pin": null,
          "type": "paper",
          "url": "https://doi.org/10.1371/journal.pcbi.1011953"
        },
        {
          "access_notes": "Official resource pinned immutably during intake.",
          "id": "rfah-benchmark-official-repository-resource",
          "last_checked": "2026-07-31",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://github.com/luhong88/int_seq_des/commit/b16a0ef3d5c44f65714b1a6c51826f6b4bdaa998",
            "value": "b16a0ef3d5c44f65714b1a6c51826f6b4bdaa998"
          },
          "type": "repository",
          "url": "https://github.com/luhong88/int_seq_des"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "initial-release",
        "entries": [
          {
            "confidence": "high",
            "count": null,
            "count_basis": "Creator-defined multistate protein sequence-design benchmark system.",
            "count_ref": null,
            "count_unit": "systems",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "rfah-benchmark-automated-metadata-2-evidence"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "The two modeled conformational states are objective dimensions, not two independent sequence-design tasks.",
            "reporting_status": "not_reported",
            "task_type_id": "protein-sequence-design"
          }
        ],
        "notes": "The creator paper explicitly frames this benchmark system as multistate protein sequence design; more specific downstream design objectives are not exhaustively classified.",
        "status": "partial"
      },
      "summary": "A multistate protein sequence-design benchmark using the fold-switching conformations of RfaH.",
      "task_counts": {
        "basis": "Fig. 1B explicitly defines the complete RfaH design problem as two-state.",
        "reporting_status": "reported",
        "subsets": [],
        "total": 2
      },
      "task_formats": [
        "unclassified"
      ],
      "verification": {
        "last_verified": "2026-07-31",
        "notes": "New family admitted after creator source and official resource verification; the unresolved benchmark license is visibly flagged.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": null,
          "evidence_ids": [
            "rfah-benchmark-automated-count-evidence",
            "rfah-benchmark-automated-version-evidence"
          ],
          "formal_tracks": [],
          "id": "rfah-benchmark-initial-release-version",
          "label": "initial-release",
          "notes": "Atlas snapshot label for the creator-paper release; the source does not report a formal benchmark version.",
          "release_date": "2024-07-11",
          "status": "current",
          "task_counts": {
            "basis": "Fig. 1B explicitly defines the complete RfaH design problem as two-state.",
            "reporting_status": "reported",
            "subsets": [],
            "total": 2
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "Aggregate result breakdowns and trajectories for released evaluations are available.",
        "biosafety_notes": null,
        "grader": "Deterministic graders, linter, and agent harness are available in the official repository.",
        "level": "partially-open",
        "license": "Apache-2.0",
        "tasks": "Canonical evaluation specifications are public; the full suite is withheld to prevent training contamination."
      },
      "aliases": [
        "SingleCellBench"
      ],
      "audit": {
        "audited_date": "2026-07-28",
        "notes": "The current 195-evaluation repository snapshot is pinned separately from the superseded 394-question creator-paper version; the latter retains its owner-reviewed subcount conflict caveat.",
        "status": "audited-with-caveats",
        "unresolved_fields": 1
      },
      "benchmark_use_ids": [
        "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an-scbench-1-use",
        "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an-scbench-2-use",
        "system-card-claude-opus-5-scbench-3-use"
      ],
      "capabilities": [
        "data-analysis",
        "coding",
        "tool-use",
        "scientific-reasoning"
      ],
      "coverage_notes": [],
      "domains": [
        "life-science",
        "transcriptomics",
        "single-cell",
        "bioinformatics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an",
        "system-card-claude-opus-5"
      ],
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-27",
          "id": "scbench-automated-metadata-evidence",
          "locator": {
            "document_page": 1,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "bb81e0c838e519898cd8437921990a04162d365b1253bcdb86c43c2bd3fdb9de",
            "type": "section",
            "value": "Abstract; Sections 4.1–5"
          },
          "source_id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/organizations",
            "/release_date",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/access/level",
            "/access/license"
          ]
        },
        {
          "accessed_date": "2026-07-27",
          "id": "scbench-automated-count-evidence",
          "locator": {
            "document_page": 2,
            "note": null,
            "printed_page": "2",
            "source_fragment_sha256": "b02c35d4261b5d98b227145422021bb79d7453ee9d29b0b1ac7907de6f76a6ce",
            "type": "table",
            "value": "Table 1"
          },
          "source_id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an",
          "source_type": "work",
          "supports": [
            "/versions/0/task_counts"
          ]
        },
        {
          "accessed_date": "2026-07-27",
          "id": "scbench-automated-resource-evidence",
          "locator": {
            "document_page": 10,
            "note": null,
            "printed_page": "10",
            "source_fragment_sha256": "e2048a14b94bcfca6fd058805209f11488b3438e2e994d8646ae2a24429b64b5",
            "type": "section",
            "value": "Section 5"
          },
          "source_id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an",
          "source_type": "work",
          "supports": [
            "/resources",
            "/implementations"
          ]
        },
        {
          "accessed_date": "2026-07-27",
          "id": "scbench-automated-creator-evidence",
          "locator": {
            "document_page": 1,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "2b4a5d08265bf2f93cee8e3e2036194670b8aa5b8b6ea1a6ecdf7ac1a4b10fca",
            "type": "page",
            "value": "Title page"
          },
          "source_id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an",
          "source_type": "work",
          "supports": [
            "/resources/0"
          ]
        },
        {
          "accessed_date": "2026-07-27",
          "id": "scbench-automated-version-evidence",
          "locator": {
            "document_page": 1,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "2b4a5d08265bf2f93cee8e3e2036194670b8aa5b8b6ea1a6ecdf7ac1a4b10fca",
            "type": "page",
            "value": "Title page"
          },
          "source_id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an",
          "source_type": "work",
          "supports": [
            "/latest_version",
            "/versions/0"
          ]
        },
        {
          "accessed_date": "2026-07-27",
          "id": "scbench-automated-count-conflict-evidence",
          "locator": {
            "document_page": 13,
            "note": null,
            "printed_page": "13",
            "source_fragment_sha256": "fac57b1733596d9eb7a0e74770c7b8985f840fe16a690a48f0cec95234899f75",
            "type": "table",
            "value": "Table 7 (continued)"
          },
          "source_id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an",
          "source_type": "work",
          "supports": [
            "/versions/0/task_counts/subsets"
          ]
        },
        {
          "accessed_date": "2026-07-28",
          "id": "scbench-repository-195-evidence",
          "locator": {
            "document_page": null,
            "note": "Pinned official README reports 195 evaluations and describes the current withheld benchmark snapshot.",
            "printed_page": null,
            "source_fragment_sha256": "c9994db5af0ce06ae840b5b55a418c4e75e5bc17a9a6fbfda4772f384557d1d9",
            "type": "repository-path",
            "value": "README.md#benchmark-structure"
          },
          "source_id": "scbench-official-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/versions/1"
          ]
        },
        {
          "accessed_date": "2026-07-28",
          "id": "scbench-repository-license-evidence",
          "locator": {
            "document_page": null,
            "note": "Pinned official README states the repository license.",
            "printed_page": null,
            "source_fragment_sha256": "cf989ea142f203f532233fa944ee1d8d5acfe6f6875e4d611f515ee108ba26df",
            "type": "repository-path",
            "value": "README.md#license"
          },
          "source_id": "scbench-official-repository-resource",
          "source_type": "resource",
          "supports": [
            "/access/license"
          ]
        },
        {
          "accessed_date": "2026-07-28",
          "id": "scbench-singlecellbench-provider-alias-evidence",
          "locator": {
            "document_page": 188,
            "note": "Provider-used evaluation label; not represented as a creator-preferred benchmark name.",
            "printed_page": "188",
            "source_fragment_sha256": "47cd5e22fa724cb5b4b57941fdf618750c717cee0eb0e3577eddaddaad53f201",
            "type": "section",
            "value": "Section 8.17.2, LatchBio Bioinformatics"
          },
          "source_id": "scbench-anthropic-opus-5-system-card-resource",
          "source_type": "resource",
          "supports": [
            "/aliases/0"
          ]
        }
      ],
      "field_status": [
        {
          "confidence": "high",
          "evidence_ids": [
            "scbench-automated-count-conflict-evidence"
          ],
          "path": "/versions/0/task_counts/subsets",
          "reason": "The owner approved the independently supported root total while all conflicted inventory subcounts were excluded from publication.",
          "status": "conflicted"
        }
      ],
      "id": "scbench",
      "implementations": [
        {
          "commit": "0bc34032bfa402dad29fc40b4cf10ea8fc03193e",
          "framework": "official repository",
          "notes": "Commit resolved deterministically during paper intake.",
          "status": "official",
          "url": "https://github.com/latchbio/scbench"
        }
      ],
      "kind": "agentic-eval",
      "latest_version": "repository-195-evaluations",
      "modalities": [
        "text",
        "raw-omics",
        "code"
      ],
      "name": "scBench",
      "organizations": [
        "LatchBio"
      ],
      "parent_id": null,
      "release_date": "2026-02-09",
      "resources": [
        {
          "access_notes": "Versioned creator source; full text is not mirrored.",
          "id": "scbench-creator-paper-resource",
          "last_checked": "2026-07-27",
          "license": null,
          "pin": null,
          "type": "paper",
          "url": "https://arxiv.org/abs/2602.09063"
        },
        {
          "access_notes": "Official repository pinned during intake.",
          "id": "scbench-official-repository-resource",
          "last_checked": "2026-07-28",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://github.com/latchbio/scbench/commit/0bc34032bfa402dad29fc40b4cf10ea8fc03193e",
            "value": "0bc34032bfa402dad29fc40b4cf10ea8fc03193e"
          },
          "type": "repository",
          "url": "https://github.com/latchbio/scbench"
        },
        {
          "access_notes": "Official model-provider source using SingleCellBench for LatchBio's 195-problem single-cell evaluation variant.",
          "id": "scbench-anthropic-opus-5-system-card-resource",
          "last_checked": "2026-07-28",
          "license": null,
          "pin": {
            "kind": "snapshot",
            "url": "https://www-cdn.anthropic.com/b514064af1408018e64b1ad24e7d5e75850b4ffd/Claude%20Opus%205%20System%20Card.pdf",
            "value": "sha256:897768f0f6f1724f3109279ab3f6458c9fbf496b56d5d2be14cab3a4f91ca472"
          },
          "type": "paper",
          "url": "https://www-cdn.anthropic.com/b514064af1408018e64b1ad24e7d5e75850b4ffd/Claude%20Opus%205%20System%20Card.pdf"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "repository-195-evaluations",
        "entries": [],
        "notes": "No Scientific Task claim passed independent high-confidence verification; task mapping remains pending a targeted official-source audit.",
        "status": "partial"
      },
      "summary": "Agentic evaluation suite for data-grounded single-cell analysis across diverse sequencing technologies and workflow stages.",
      "task_counts": {
        "basis": "Official repository README at commit 0bc34032bfa402dad29fc40b4cf10ea8fc03193e",
        "reporting_status": "reported",
        "subsets": [],
        "total": 195
      },
      "task_formats": [
        "natural-language data-analysis task with structured JSON output"
      ],
      "verification": {
        "last_verified": "2026-07-28",
        "notes": "Creator-paper and current official-repository inventories are versioned separately; the original appendix inventory caveat remains attached to the superseded creator-paper version.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": null,
          "evidence_ids": [
            "scbench-automated-count-evidence",
            "scbench-automated-version-evidence"
          ],
          "formal_tracks": [],
          "id": "scbench-initial-release-version",
          "label": "initial-release",
          "notes": "Root total retained after owner review; conflicted appendix inventory subcounts are intentionally omitted.",
          "release_date": "2026-02-09",
          "status": "superseded",
          "task_counts": {
            "basis": "Explicit overall benchmark total in the abstract and Table 1",
            "reporting_status": "reported",
            "subsets": [],
            "total": 394
          }
        },
        {
          "as_of": "2026-06-10",
          "evidence_ids": [
            "scbench-repository-195-evidence"
          ],
          "formal_tracks": [],
          "id": "scbench-repository-195-version",
          "label": "repository-195-evaluations",
          "notes": "Current official repository snapshot; kept separate from the 394-question creator-paper inventory.",
          "release_date": "2026-06-10",
          "status": "current",
          "task_counts": {
            "basis": "Official repository README at commit 0bc34032bfa402dad29fc40b4cf10ea8fc03193e",
            "reporting_status": "reported",
            "subsets": [],
            "total": 195
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The scIB package, Snakemake benchmarking pipeline, metrics, and reproducibility website are public.",
        "biosafety_notes": "No biosafety-specific restrictions are identified; the registry mirrors no cell-level data.",
        "grader": "Fourteen deterministic metrics plus batch-removal, bio-conservation, and 60/40 weighted overall scores.",
        "level": "fully-open",
        "license": "MIT for scIB code; source datasets from 23 publications retain their original terms.",
        "tasks": "Preprocessed integration tasks, result matrices, and source datasets are linked from the reproducibility project."
      },
      "aliases": [
        "Single-cell integration benchmarking",
        "Benchmarking atlas-level data integration"
      ],
      "audit": {
        "audited_date": "2026-07-22",
        "notes": "Task composition, 16 methods, four preprocessing combinations, 14 metrics, and weighted aggregation were checked against the paper.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "data-analysis"
      ],
      "coverage_notes": [
        {
          "count": 13,
          "coverage": "explicitly-in-scope",
          "notes": "All thirteen tasks assess integration of single-cell datasets.",
          "reporting_status": "reported",
          "tag": "single-cell"
        },
        {
          "count": 5,
          "coverage": "explicitly-in-scope",
          "notes": "Five real scRNA-seq integration tasks; simulation is counted separately.",
          "reporting_status": "reported",
          "tag": "transcriptomics"
        },
        {
          "count": 6,
          "coverage": "explicitly-in-scope",
          "notes": "Six scATAC-seq integration tasks.",
          "reporting_status": "reported",
          "tag": "epigenomics"
        }
      ],
      "domains": [
        "single-cell",
        "transcriptomics",
        "epigenomics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "scib-paper"
      ],
      "evaluation_run_ids": [
        "scib-creator-full"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "scib-paper-definition-evidence",
          "locator": {
            "note": "Reports 13 tasks as 2 simulation, 5 scRNA-seq, and 6 scATAC-seq; 16 methods, 4 preprocessing combinations, 14 metrics, and 0.6 bio plus 0.4 batch aggregation.",
            "type": "section",
            "value": "Abstract; Results pp. 41-50, Figure 1, Table 1, Methods: Metric aggregation"
          },
          "source_id": "scib-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/resources",
            "/versions/0/release_date",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0"
          ]
        },
        {
          "accessed_date": "2026-07-22",
          "id": "scib-repository-evidence",
          "locator": {
            "note": "Pins public pipeline, task artifacts, metrics implementation, and code licenses.",
            "type": "repository-path",
            "value": "README.md and workflow/config sources at 3afbffd3674726e5146797be21cf6bd7470a2c5f; scib package at cd67913396b4c0430710b3d90f1d1841f5fa4468"
          },
          "source_id": "scib-reproducibility-resource",
          "source_type": "resource",
          "supports": [
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/implementations",
            "/versions/0/formal_tracks"
          ]
        }
      ],
      "field_status": [],
      "id": "scib",
      "implementations": [
        {
          "commit": "cd67913396b4c0430710b3d90f1d1841f5fa4468",
          "framework": "scIB package",
          "notes": "Integration wrappers and metric implementation.",
          "status": "official",
          "url": "https://github.com/theislab/scib/tree/cd67913396b4c0430710b3d90f1d1841f5fa4468"
        },
        {
          "commit": "3afbffd3674726e5146797be21cf6bd7470a2c5f",
          "framework": "scIB reproducibility pipeline",
          "notes": "Paper-specific tasks, preprocessing combinations, workflow, and result generation.",
          "status": "official",
          "url": "https://github.com/theislab/scib-reproducibility/tree/3afbffd3674726e5146797be21cf6bd7470a2c5f"
        }
      ],
      "kind": "suite",
      "latest_version": "paper-2021",
      "modalities": [
        "raw-omics",
        "table"
      ],
      "name": "scIB",
      "organizations": [
        "Helmholtz Zentrum München",
        "Technical University of Munich"
      ],
      "parent_id": null,
      "release_date": "2021-12-23",
      "resources": [
        {
          "access_notes": "Peer-reviewed creator paper in Nature Methods.",
          "id": "scib-paper-resource",
          "last_checked": "2026-07-22",
          "license": "CC BY 4.0",
          "pin": null,
          "type": "paper",
          "url": "https://doi.org/10.1038/s41592-021-01336-8"
        },
        {
          "access_notes": "Official package implementing integration wrappers and metrics.",
          "id": "scib-repository-resource",
          "last_checked": "2026-07-22",
          "license": "MIT",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/theislab/scib/tree/cd67913396b4c0430710b3d90f1d1841f5fa4468",
            "value": "cd67913396b4c0430710b3d90f1d1841f5fa4468"
          },
          "type": "repository",
          "url": "https://github.com/theislab/scib"
        },
        {
          "access_notes": "Paper-specific pipeline and reproducibility source.",
          "id": "scib-reproducibility-resource",
          "last_checked": "2026-07-22",
          "license": "MIT",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/theislab/scib-reproducibility/tree/3afbffd3674726e5146797be21cf6bd7470a2c5f",
            "value": "3afbffd3674726e5146797be21cf6bd7470a2c5f"
          },
          "type": "repository",
          "url": "https://github.com/theislab/scib-reproducibility"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "paper-2021",
        "entries": [
          {
            "confidence": "high",
            "count": 13,
            "count_basis": "Atlas-level integration tasks.",
            "count_ref": "/task_counts/total",
            "count_unit": "tasks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "scib-paper-definition-evidence"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "Two simulation, five scRNA-seq, and six scATAC-seq integration tasks.",
            "reporting_status": "reported",
            "task_type_id": "single-cell-data-integration"
          }
        ],
        "notes": "The creator study evaluates integration methods across thirteen atlas-level tasks and does not equate cells or batches with task count.",
        "status": "complete"
      },
      "summary": "A 13-task benchmark of single-cell data integration across simulated, scRNA-seq, and scATAC-seq settings, evaluated with 14 batch-removal and biological-conservation metrics.",
      "task_counts": {
        "basis": "atlas-level integration tasks",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "atlas-level integration tasks",
            "count": 2,
            "exclusive": true,
            "exhaustive": true,
            "id": "scib-simulation",
            "label": "Simulation tasks",
            "notes": "Two simulations with known ground truth.",
            "reporting_status": "reported"
          },
          {
            "basis": "atlas-level integration tasks",
            "count": 5,
            "exclusive": true,
            "exhaustive": true,
            "id": "scib-scrna",
            "label": "scRNA-seq tasks",
            "notes": "Five RNA integration settings.",
            "reporting_status": "reported"
          },
          {
            "basis": "atlas-level integration tasks",
            "count": 6,
            "exclusive": true,
            "exhaustive": true,
            "id": "scib-scatac",
            "label": "scATAC-seq tasks",
            "notes": "Six ATAC integration settings across feature spaces and scales.",
            "reporting_status": "reported"
          }
        ],
        "total": 13
      },
      "task_formats": [
        "single-cell batch integration",
        "representation evaluation"
      ],
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Paper and paper-specific reproducibility pipeline define the registered snapshot.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": null,
          "evidence_ids": [
            "scib-paper-definition-evidence",
            "scib-repository-evidence"
          ],
          "formal_tracks": [],
          "id": "scib-paper-2021",
          "label": "paper-2021",
          "notes": "Fixed paper snapshot of thirteen integration tasks; later scIB/scib-metrics changes are not silently folded into this version.",
          "release_date": "2021-12-23",
          "status": "current",
          "task_counts": {
            "basis": "atlas-level integration tasks",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "atlas-level integration tasks",
                "count": 2,
                "exclusive": true,
                "exhaustive": true,
                "id": "scib-simulation",
                "label": "Simulation tasks",
                "notes": "Two simulations with known ground truth.",
                "reporting_status": "reported"
              },
              {
                "basis": "atlas-level integration tasks",
                "count": 5,
                "exclusive": true,
                "exhaustive": true,
                "id": "scib-scrna",
                "label": "scRNA-seq tasks",
                "notes": "Five RNA integration settings.",
                "reporting_status": "reported"
              },
              {
                "basis": "atlas-level integration tasks",
                "count": 6,
                "exclusive": true,
                "exhaustive": true,
                "id": "scib-scatac",
                "label": "scATAC-seq tasks",
                "notes": "Six ATAC integration settings across feature spaces and scales.",
                "reporting_status": "reported"
              }
            ],
            "total": 13
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The creators publish the benchmark data, 2,466 main-evaluation trajectories/final models, prompts, agent environment, and metric code; the repositories and dataset cards do not state redistribution licenses.",
        "biosafety_notes": "SCIGYM exposes de-identified computational models and simulated concentration trajectories, not live materials, pathogen sequences, or wet-lab execution instructions; no separate creator biosafety restriction is reported.",
        "grader": "Public deterministic code computes network-topology F1, reaction-matching precision/recall/F1 with and without modifiers, and SMAPE-based simulation trajectory error.",
        "level": "fully-open",
        "license": null,
        "tasks": "All 350 partial/reference SBML systems and SED-ML simulation configurations are publicly downloadable in 137-system small and 213-system large splits."
      },
      "aliases": [
        "SciGym SBML"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Identity, release history, split rows, task protocol, exact model versions, public trajectories, metrics, and Table 1 results were audited. The code and repackaged benchmark data are public but neither upstream source states a license.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "experiment-planning",
        "data-analysis",
        "coding",
        "tool-use",
        "scientific-reasoning"
      ],
      "coverage_notes": [
        {
          "count": null,
          "coverage": "explicitly-in-scope",
          "notes": "The source systems include signaling, metabolic, regulatory, and other biological processes, but the creators do not publish a mutually exclusive molecular/cell-biology count.",
          "reporting_status": "not_reported",
          "tag": "molecular-cell-biology"
        },
        {
          "count": null,
          "coverage": "unknown",
          "notes": "Gene-regulatory networks are named as an example system class, but no official task-level genomics count is published and the benchmark does not use genomic sequence data.",
          "reporting_status": "not_reported",
          "tag": "genomics"
        },
        {
          "count": 0,
          "coverage": "not-in-scope",
          "notes": "The paper motivates the task by analogy to Perturb-seq and spatial transcriptomics, but SCIGYM inputs are SBML systems and simulated concentration trajectories rather than transcriptomic measurements.",
          "reporting_status": "reported",
          "tag": "transcriptomics"
        },
        {
          "count": 0,
          "coverage": "not-in-scope",
          "notes": "No single-cell measurement is an official benchmark input or target.",
          "reporting_status": "reported",
          "tag": "single-cell"
        },
        {
          "count": 0,
          "coverage": "not-in-scope",
          "notes": "Agents reconstruct reaction networks; they do not design or optimize protein sequences or structures.",
          "reporting_status": "reported",
          "tag": "protein-design"
        },
        {
          "count": 0,
          "coverage": "not-in-scope",
          "notes": "Species can represent proteins, but no task evaluates protein-protein binding prediction or affinity.",
          "reporting_status": "reported",
          "tag": "protein-protein-binding"
        },
        {
          "count": 0,
          "coverage": "not-in-scope",
          "notes": "No task evaluates protein-ligand binding prediction or affinity.",
          "reporting_status": "reported",
          "tag": "protein-ligand-binding"
        }
      ],
      "domains": [
        "life-science",
        "molecular-cell-biology",
        "bioinformatics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "scigym-evidence-identity",
          "locator": {
            "note": "Defines SCIGYM, all five represented organizations, end-to-end simulated discovery, SBML inputs, ReAct agent actions, and the reconstruction task.",
            "type": "page",
            "value": "NeurIPS paper pp. 1–5, title, author affiliations, abstract, and §§1–3.2"
          },
          "source_id": "scigym-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/organizations",
            "/kind",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "scigym-evidence-release-counts",
          "locator": {
            "note": "The first public release commit is dated 2025-05-16; the files contain 137 and 213 unique system rows, totaling 350.",
            "type": "dataset-card",
            "value": "README.md and data/{small,large}-00000-of-00001.parquet at commit dbb10c12a33427e8e05ebbac66de690c65b6acae"
          },
          "source_id": "scigym-dataset-resource",
          "source_type": "resource",
          "supports": [
            "/release_date",
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/versions/0/task_counts",
            "/versions/0/formal_tracks",
            "/scientific_task_classification/entries/0",
            "/scientific_task_classification/entries/1"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "scigym-evidence-taxonomy",
          "locator": {
            "note": "Describes signaling, metabolic, regulatory, and epidemiological system classes; the released tasks use SBML and simulated time series, not sequence, omics, binding, or protein-design targets.",
            "type": "section",
            "value": "NeurIPS paper §§1–3, 5–6 and Appendix C.3"
          },
          "source_id": "scigym-paper",
          "source_type": "work",
          "supports": [
            "/domains",
            "/coverage_notes",
            "/access/biosafety_notes",
            "/scientific_task_classification/entries/0",
            "/scientific_task_classification/entries/1"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "scigym-evidence-access-license",
          "locator": {
            "note": "The environment, prompts, and grader are public, but no software license is stated. The two pinned Hugging Face cards likewise have no license field.",
            "type": "repository-path",
            "value": "Complete tree, README.md, pyproject.toml, scigym/prompts/, scigym/eval/, and absence of LICENSE at commit 88a7b93609e35b6ecb4eb343d816d6ff09256c6a"
          },
          "source_id": "scigym-repository-resource",
          "source_type": "resource",
          "supports": [
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/implementations"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "scigym-evidence-results-artifact",
          "locator": {
            "note": "2,466 public rows comprise 137 systems × 6 exact models × 3 episodes; each row includes chat_history and final_model.",
            "type": "dataset-card",
            "value": "data/small-00000-of-00001.parquet at commit 7d472c12855d46702c4915892578290355894c1a"
          },
          "source_id": "scigym-results-resource",
          "source_type": "resource",
          "supports": [
            "/access/artifacts"
          ]
        }
      ],
      "field_status": [],
      "id": "scigym",
      "implementations": [
        {
          "commit": "88a7b93609e35b6ecb4eb343d816d6ff09256c6a",
          "framework": "SCIGYM Python package 0.0.1",
          "notes": "Python 3.10 environment with Tellurium/libRoadRunner, libSBML, public prompts, model adapters, and deterministic evaluator; upstream supplies no software license.",
          "status": "official",
          "url": "https://github.com/h4duan/SciGym/tree/88a7b93609e35b6ecb4eb343d816d6ff09256c6a"
        },
        {
          "commit": "d290bb04bf54aad1c473c4701e0d0d88013c4f91",
          "framework": "SCIGYM NeurIPS evaluation harness",
          "notes": "Evaluation-era controller and exact six-model API adapters used with the public trajectory release; upstream supplies no software license.",
          "status": "official",
          "url": "https://github.com/h4duan/scigym-neurips/tree/d290bb04bf54aad1c473c4701e0d0d88013c4f91"
        }
      ],
      "kind": "suite",
      "latest_version": "2025 release",
      "modalities": [
        "text",
        "table",
        "code"
      ],
      "name": "SCIGYM",
      "organizations": [
        "University of Toronto",
        "SickKids",
        "Axiom",
        "Mila",
        "Vector Institute"
      ],
      "parent_id": null,
      "release_date": "2025-05-16",
      "resources": [
        {
          "access_notes": "Final NeurIPS 2025 Datasets and Benchmarks Track creator paper with the complete appendix.",
          "id": "scigym-paper-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": null,
          "type": "paper",
          "url": "https://proceedings.neurips.cc/paper_files/paper/2025/hash/a23760ba51036d530a2656c3835f826c-Abstract-Datasets_and_Benchmarks_Track.html"
        },
        {
          "access_notes": "Official project page; captured HTML SHA256 a4594c7b165676f6f9b9c9bfb89379339355f36ec91006e1ab9a557eca013e17.",
          "id": "scigym-website-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://github.com/h4duan/scigym-benchmark/tree/5af9c5d542cdda3640706efebb4ffd46f6416f94",
            "value": "5af9c5d542cdda3640706efebb4ffd46f6416f94"
          },
          "type": "website",
          "url": "https://h4duan.github.io/scigym-benchmark/"
        },
        {
          "access_notes": "Current official Python package 0.0.1, prompts, environment, evaluator, and download scripts; the one-commit repository has no LICENSE file or releases.",
          "id": "scigym-repository-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://github.com/h4duan/SciGym/tree/88a7b93609e35b6ecb4eb343d816d6ff09256c6a",
            "value": "88a7b93609e35b6ecb4eb343d816d6ff09256c6a"
          },
          "type": "repository",
          "url": "https://github.com/h4duan/SciGym"
        },
        {
          "access_notes": "Creator's evaluation-era implementation with all six exact API model identifiers, main and ablation prompts, simulator, controller, and grader; no LICENSE file is present.",
          "id": "scigym-evaluation-code-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://github.com/h4duan/scigym-neurips/tree/d290bb04bf54aad1c473c4701e0d0d88013c4f91",
            "value": "d290bb04bf54aad1c473c4701e0d0d88013c4f91"
          },
          "type": "repository",
          "url": "https://github.com/h4duan/scigym-neurips"
        },
        {
          "access_notes": "Official 350-row release. The pinned Parquet files contain 137 unique small and 213 unique large folder IDs; SHA256 values are 7e7e33afd953a6eedb8a43548c4a41ab1b9077b5c7f729484aae6a9aa41e5b76 and ef4a0f53a23d40579f4b23308f8def7164ae6a20ea05acb80fad4afdc8673fe9. The card declares no license.",
          "id": "scigym-dataset-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://huggingface.co/datasets/h4duan/scigym-sbml/tree/dbb10c12a33427e8e05ebbac66de690c65b6acae",
            "value": "dbb10c12a33427e8e05ebbac66de690c65b6acae"
          },
          "type": "dataset",
          "url": "https://huggingface.co/datasets/h4duan/scigym-sbml"
        },
        {
          "access_notes": "Official 2,466-row main-evaluation artifact containing chat histories and final SBML models. It covers 137 systems, six exact model identifiers, and three rows per model-system pair; Parquet SHA256 94750cff79e142a0a5142f1abc8d827b104338b7337e082bde641fb05ee83177. The card is empty and declares no license.",
          "id": "scigym-results-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://huggingface.co/datasets/h4duan/scigym-eval/tree/7d472c12855d46702c4915892578290355894c1a",
            "value": "7d472c12855d46702c4915892578290355894c1a"
          },
          "type": "dataset",
          "url": "https://huggingface.co/datasets/h4duan/scigym-eval"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "2025 release",
        "entries": [
          {
            "confidence": "high",
            "count": 350,
            "count_basis": "distinct curated BioModels systems released as SBML benchmark instances",
            "count_ref": "/task_counts/total",
            "count_unit": "systems",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "scigym-evidence-release-counts",
              "scigym-evidence-taxonomy"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "Hidden biological reactions are reconstructed from interventions and trajectories.",
            "reporting_status": "reported",
            "task_type_id": "reaction-network-reconstruction"
          },
          {
            "confidence": "high",
            "count": 350,
            "count_basis": "distinct curated BioModels systems released as SBML benchmark instances",
            "count_ref": "/task_counts/total",
            "count_unit": "systems",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "scigym-evidence-release-counts",
              "scigym-evidence-taxonomy"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "The same systems support iterative in-silico perturbation experiments; overlapping claims are not summed.",
            "reporting_status": "reported",
            "task_type_id": "simulation-based-experiment"
          }
        ],
        "notes": "All released systems use the same simulator-based reaction-network reconstruction task.",
        "status": "complete"
      },
      "summary": "An agentic systems-biology suite in which language models iteratively perturb simulated SBML systems, analyze time-series observations in Python, and reconstruct hidden biological reactions.",
      "task_counts": {
        "basis": "distinct curated BioModels systems released as SBML benchmark instances",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "systems with fewer than 10 reactions in the evaluated small split",
            "count": 137,
            "exclusive": true,
            "exhaustive": true,
            "id": "scigym-small-systems",
            "label": "Small systems",
            "notes": "All six creator-paper models were evaluated on this split.",
            "reporting_status": "reported"
          },
          {
            "basis": "remaining released systems with up to 400 reactions in the large split",
            "count": 213,
            "exclusive": true,
            "exhaustive": true,
            "id": "scigym-large-systems",
            "label": "Large systems",
            "notes": "Released by the creators but not evaluated in the paper.",
            "reporting_status": "reported"
          }
        ],
        "total": 350
      },
      "task_formats": [
        "interactive simulated experiment",
        "SBML reaction-network reconstruction"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Audited against the final NeurIPS paper, official website, two commit-pinned creator code repositories, and immutable benchmark/evaluation data snapshots.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "scigym-evidence-release-counts"
          ],
          "formal_tracks": [
            "scigym-small",
            "scigym-large"
          ],
          "id": "scigym-2025-release",
          "label": "2025 release",
          "notes": "First public code/data release. The small and large Parquet splits remain at 137 and 213 systems in their pinned official repositories, and no later version or tag has been published.",
          "release_date": "2025-05-16",
          "status": "current",
          "task_counts": {
            "basis": "distinct curated BioModels systems released as SBML benchmark instances",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "systems with fewer than 10 reactions in the evaluated small split",
                "count": 137,
                "exclusive": true,
                "exhaustive": true,
                "id": "scigym-small-systems",
                "label": "Small systems",
                "notes": "All six creator-paper models were evaluated on this split.",
                "reporting_status": "reported"
              },
              {
                "basis": "remaining released systems with up to 400 reactions in the large split",
                "count": 213,
                "exclusive": true,
                "exhaustive": true,
                "id": "scigym-large-systems",
                "label": "Large systems",
                "notes": "Released by the creators but not evaluated in the paper.",
                "reporting_status": "reported"
              }
            ],
            "total": 350
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The released track includes reference and partial systems but no creator model trajectories; upstream does not state a data license.",
        "biosafety_notes": "This track contains de-identified computational models and simulated outputs, with no live materials or wet-lab execution.",
        "grader": "The public SCIGYM deterministic evaluator supports this track, although the creator paper did not run model evaluations on it.",
        "level": "fully-open",
        "license": null,
        "tasks": "All 213 partial/reference SBML and SED-ML instances are public."
      },
      "aliases": [
        "SCIGYM large split"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Count, access, runnable path, and non-evaluated status were audited; code and data licenses remain unreported upstream.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "experiment-planning",
        "data-analysis",
        "coding",
        "tool-use",
        "scientific-reasoning"
      ],
      "coverage_notes": [
        {
          "count": null,
          "coverage": "unknown",
          "notes": "The official split does not publish a task-level genomics classification.",
          "reporting_status": "not_reported",
          "tag": "genomics"
        },
        {
          "count": 0,
          "coverage": "not-in-scope",
          "notes": "Tasks consume SBML and simulated trajectories rather than transcriptomic measurements.",
          "reporting_status": "reported",
          "tag": "transcriptomics"
        },
        {
          "count": 0,
          "coverage": "not-in-scope",
          "notes": "The task reconstructs reaction networks rather than proteins.",
          "reporting_status": "reported",
          "tag": "protein-design"
        },
        {
          "count": 0,
          "coverage": "not-in-scope",
          "notes": "No task evaluates protein-protein binding.",
          "reporting_status": "reported",
          "tag": "protein-protein-binding"
        },
        {
          "count": 0,
          "coverage": "not-in-scope",
          "notes": "No task evaluates protein-ligand binding.",
          "reporting_status": "reported",
          "tag": "protein-ligand-binding"
        }
      ],
      "domains": [
        "life-science",
        "molecular-cell-biology",
        "bioinformatics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "scigym-large-evidence-count",
          "locator": {
            "note": "213 unique large-split systems; the paper describes them as the remaining released systems with up to 400 reactions and says they were not evaluated.",
            "type": "dataset-card",
            "value": "README.md and data/large-00000-of-00001.parquet at commit dbb10c12a33427e8e05ebbac66de690c65b6acae"
          },
          "source_id": "scigym-large-dataset-resource",
          "source_type": "resource",
          "supports": [
            "/name",
            "/summary",
            "/kind",
            "/parent_id",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/versions/0/task_counts",
            "/scientific_task_classification/entries/0",
            "/scientific_task_classification/entries/1"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "scigym-large-evidence-access",
          "locator": {
            "note": "Public large-split download path, environment, and grader with no stated software or data license.",
            "type": "repository-path",
            "value": "README.md, data/download.py, scigym/, and absence of LICENSE at commit 88a7b93609e35b6ecb4eb343d816d6ff09256c6a"
          },
          "source_id": "scigym-large-repository-resource",
          "source_type": "resource",
          "supports": [
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/resources",
            "/implementations",
            "/coverage_notes"
          ]
        }
      ],
      "field_status": [],
      "id": "scigym-large",
      "implementations": [
        {
          "commit": "88a7b93609e35b6ecb4eb343d816d6ff09256c6a",
          "framework": "SCIGYM Python package 0.0.1",
          "notes": "The same public environment and grader can load the large split; the creators report no paper evaluation for it.",
          "status": "official",
          "url": "https://github.com/h4duan/SciGym/tree/88a7b93609e35b6ecb4eb343d816d6ff09256c6a"
        }
      ],
      "kind": "track",
      "latest_version": "2025 release",
      "modalities": [
        "text",
        "table",
        "code"
      ],
      "name": "SCIGYM Large",
      "organizations": [
        "University of Toronto",
        "SickKids",
        "Axiom",
        "Mila",
        "Vector Institute"
      ],
      "parent_id": "scigym",
      "release_date": "2025-05-16",
      "resources": [
        {
          "access_notes": "Final creator paper and appendix.",
          "id": "scigym-large-paper-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": null,
          "type": "paper",
          "url": "https://proceedings.neurips.cc/paper_files/paper/2025/hash/a23760ba51036d530a2656c3835f826c-Abstract-Datasets_and_Benchmarks_Track.html"
        },
        {
          "access_notes": "Official 213-row large split; Parquet SHA256 ef4a0f53a23d40579f4b23308f8def7164ae6a20ea05acb80fad4afdc8673fe9.",
          "id": "scigym-large-dataset-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://huggingface.co/datasets/h4duan/scigym-sbml/tree/dbb10c12a33427e8e05ebbac66de690c65b6acae",
            "value": "dbb10c12a33427e8e05ebbac66de690c65b6acae"
          },
          "type": "dataset",
          "url": "https://huggingface.co/datasets/h4duan/scigym-sbml"
        },
        {
          "access_notes": "Current public runner and grader; no LICENSE file.",
          "id": "scigym-large-repository-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://github.com/h4duan/SciGym/tree/88a7b93609e35b6ecb4eb343d816d6ff09256c6a",
            "value": "88a7b93609e35b6ecb4eb343d816d6ff09256c6a"
          },
          "type": "repository",
          "url": "https://github.com/h4duan/SciGym"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "2025 release",
        "entries": [
          {
            "confidence": "high",
            "count": 213,
            "count_basis": "unique SBML systems in the official large Parquet split, containing the remaining systems with up to 400 reactions",
            "count_ref": "/task_counts/total",
            "count_unit": "systems",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "scigym-large-evidence-count"
            ],
            "mapping_method": "official-track",
            "notes": "Split-specific system count.",
            "reporting_status": "reported",
            "task_type_id": "reaction-network-reconstruction"
          },
          {
            "confidence": "high",
            "count": 213,
            "count_basis": "unique SBML systems in the official large Parquet split, containing the remaining systems with up to 400 reactions",
            "count_ref": "/task_counts/total",
            "count_unit": "systems",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "scigym-large-evidence-count"
            ],
            "mapping_method": "official-track",
            "notes": "Same systems; overlapping task claim.",
            "reporting_status": "reported",
            "task_type_id": "simulation-based-experiment"
          }
        ],
        "notes": "Official large-system split.",
        "status": "complete"
      },
      "summary": "The formally released SCIGYM track containing the 213 systems not included in the creator paper's model evaluation, with systems reaching up to 400 reactions.",
      "task_counts": {
        "basis": "unique SBML systems in the official large Parquet split, containing the remaining systems with up to 400 reactions",
        "reporting_status": "reported",
        "subsets": [],
        "total": 213
      },
      "task_formats": [
        "interactive simulated experiment",
        "SBML reaction-network reconstruction"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Audited against the final creator paper and immutable official dataset/repository snapshots.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "scigym-large-evidence-count"
          ],
          "formal_tracks": [],
          "id": "scigym-large-2025-release",
          "label": "2025 release",
          "notes": "Current 213-system large split; released but not evaluated in the creator paper.",
          "release_date": "2025-05-16",
          "status": "current",
          "task_counts": {
            "basis": "unique SBML systems in the official large Parquet split, containing the remaining systems with up to 400 reactions",
            "reporting_status": "reported",
            "subsets": [],
            "total": 213
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "All 2,466 main creator-evaluation trajectories and final models are public; upstream does not state a data license.",
        "biosafety_notes": "This track contains de-identified computational models and simulated outputs, with no live materials or wet-lab execution.",
        "grader": "Public deterministic SCIGYM evaluator computes structural and simulation metrics.",
        "level": "fully-open",
        "license": null,
        "tasks": "All 137 partial/reference SBML and SED-ML instances are public."
      },
      "aliases": [
        "SCIGYM small split"
      ],
      "audit": {
        "audited_date": "2026-07-21",
        "notes": "Count, scope, access, protocol, models, trajectories, metrics, and results were audited; code and data licenses remain unreported upstream.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "experiment-planning",
        "data-analysis",
        "coding",
        "tool-use",
        "scientific-reasoning"
      ],
      "coverage_notes": [
        {
          "count": null,
          "coverage": "unknown",
          "notes": "The official split does not publish a task-level genomics classification.",
          "reporting_status": "not_reported",
          "tag": "genomics"
        },
        {
          "count": 0,
          "coverage": "not-in-scope",
          "notes": "Tasks consume SBML and simulated trajectories rather than transcriptomic measurements.",
          "reporting_status": "reported",
          "tag": "transcriptomics"
        },
        {
          "count": 0,
          "coverage": "not-in-scope",
          "notes": "The task reconstructs reaction networks rather than proteins.",
          "reporting_status": "reported",
          "tag": "protein-design"
        },
        {
          "count": 0,
          "coverage": "not-in-scope",
          "notes": "No task evaluates protein-protein binding.",
          "reporting_status": "reported",
          "tag": "protein-protein-binding"
        },
        {
          "count": 0,
          "coverage": "not-in-scope",
          "notes": "No task evaluates protein-ligand binding.",
          "reporting_status": "reported",
          "tag": "protein-ligand-binding"
        }
      ],
      "domains": [
        "life-science",
        "molecular-cell-biology",
        "bioinformatics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "scigym-paper"
      ],
      "evaluation_run_ids": [
        "scigym-small-creator-paper",
        "scigym-small-zero-shot"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "scigym-small-evidence-count",
          "locator": {
            "note": "137 unique systems; the paper defines small as systems with fewer than ten reactions.",
            "type": "dataset-card",
            "value": "README.md and data/small-00000-of-00001.parquet at commit dbb10c12a33427e8e05ebbac66de690c65b6acae"
          },
          "source_id": "scigym-small-dataset-resource",
          "source_type": "resource",
          "supports": [
            "/name",
            "/summary",
            "/kind",
            "/parent_id",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/versions/0/task_counts",
            "/scientific_task_classification/entries/0",
            "/scientific_task_classification/entries/1"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "scigym-small-evidence-access",
          "locator": {
            "note": "Public runnable harness and grader with no stated software license; public dataset/evaluation cards also state no license.",
            "type": "repository-path",
            "value": "README.md, configs/, scigym/system_prompts/, scigym/controller.py, scigym/evaluator.py, and absence of LICENSE at commit d290bb04bf54aad1c473c4701e0d0d88013c4f91"
          },
          "source_id": "scigym-small-evaluation-code-resource",
          "source_type": "resource",
          "supports": [
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/resources",
            "/implementations",
            "/coverage_notes"
          ]
        }
      ],
      "field_status": [],
      "id": "scigym-small",
      "implementations": [
        {
          "commit": "d290bb04bf54aad1c473c4701e0d0d88013c4f91",
          "framework": "SCIGYM NeurIPS evaluation harness",
          "notes": "Public small-track runner, simulator, agent loop, prompts, and evaluator; upstream license not reported.",
          "status": "official",
          "url": "https://github.com/h4duan/scigym-neurips/tree/d290bb04bf54aad1c473c4701e0d0d88013c4f91"
        }
      ],
      "kind": "track",
      "latest_version": "2025 release",
      "modalities": [
        "text",
        "table",
        "code"
      ],
      "name": "SCIGYM Small",
      "organizations": [
        "University of Toronto",
        "SickKids",
        "Axiom",
        "Mila",
        "Vector Institute"
      ],
      "parent_id": "scigym",
      "release_date": "2025-05-16",
      "resources": [
        {
          "access_notes": "Final creator paper and appendix.",
          "id": "scigym-small-paper-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": null,
          "type": "paper",
          "url": "https://proceedings.neurips.cc/paper_files/paper/2025/hash/a23760ba51036d530a2656c3835f826c-Abstract-Datasets_and_Benchmarks_Track.html"
        },
        {
          "access_notes": "Official 137-row small split; Parquet SHA256 7e7e33afd953a6eedb8a43548c4a41ab1b9077b5c7f729484aae6a9aa41e5b76.",
          "id": "scigym-small-dataset-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://huggingface.co/datasets/h4duan/scigym-sbml/tree/dbb10c12a33427e8e05ebbac66de690c65b6acae",
            "value": "dbb10c12a33427e8e05ebbac66de690c65b6acae"
          },
          "type": "dataset",
          "url": "https://huggingface.co/datasets/h4duan/scigym-sbml"
        },
        {
          "access_notes": "Official 2,466-row six-model, three-episode-per-pair evaluation artifact; Parquet SHA256 94750cff79e142a0a5142f1abc8d827b104338b7337e082bde641fb05ee83177.",
          "id": "scigym-small-results-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://huggingface.co/datasets/h4duan/scigym-eval/tree/7d472c12855d46702c4915892578290355894c1a",
            "value": "7d472c12855d46702c4915892578290355894c1a"
          },
          "type": "dataset",
          "url": "https://huggingface.co/datasets/h4duan/scigym-eval"
        },
        {
          "access_notes": "Evaluation-era environment, prompts, exact model adapters, and deterministic grader; no LICENSE file.",
          "id": "scigym-small-evaluation-code-resource",
          "last_checked": "2026-07-21",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://github.com/h4duan/scigym-neurips/tree/d290bb04bf54aad1c473c4701e0d0d88013c4f91",
            "value": "d290bb04bf54aad1c473c4701e0d0d88013c4f91"
          },
          "type": "repository",
          "url": "https://github.com/h4duan/scigym-neurips"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "2025 release",
        "entries": [
          {
            "confidence": "high",
            "count": 137,
            "count_basis": "unique SBML systems with fewer than 10 reactions in the official small Parquet split",
            "count_ref": "/task_counts/total",
            "count_unit": "systems",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "scigym-small-evidence-count"
            ],
            "mapping_method": "official-track",
            "notes": "Split-specific system count.",
            "reporting_status": "reported",
            "task_type_id": "reaction-network-reconstruction"
          },
          {
            "confidence": "high",
            "count": 137,
            "count_basis": "unique SBML systems with fewer than 10 reactions in the official small Parquet split",
            "count_ref": "/task_counts/total",
            "count_unit": "systems",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "scigym-small-evidence-count"
            ],
            "mapping_method": "official-track",
            "notes": "Same systems; overlapping task claim.",
            "reporting_status": "reported",
            "task_type_id": "simulation-based-experiment"
          }
        ],
        "notes": "Official small-system split.",
        "status": "complete"
      },
      "summary": "The formally released and creator-evaluated SCIGYM track containing biological systems with fewer than ten reactions.",
      "task_counts": {
        "basis": "unique SBML systems with fewer than 10 reactions in the official small Parquet split",
        "reporting_status": "reported",
        "subsets": [],
        "total": 137
      },
      "task_formats": [
        "interactive simulated experiment",
        "SBML reaction-network reconstruction"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Audited against the final creator paper and immutable official data/code snapshots.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-07-21",
          "evidence_ids": [
            "scigym-small-evidence-count"
          ],
          "formal_tracks": [],
          "id": "scigym-small-2025-release",
          "label": "2025 release",
          "notes": "Current 137-system small split used by the creator paper.",
          "release_date": "2025-05-16",
          "status": "current",
          "task_counts": {
            "basis": "unique SBML systems with fewer than 10 reactions in the official small Parquet split",
            "reporting_status": "reported",
            "subsets": [],
            "total": 137
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "Source code and the SOAR-RNA benchmark artifact are available through the pinned official repository; the SOAR-MultiOmics benchmark artifact was not located in that snapshot.",
        "biosafety_notes": null,
        "grader": "Not established by the independently verified metadata claims.",
        "level": "partially-open",
        "license": null,
        "tasks": "The creator paper states that the SOAR benchmark datasets are publicly available, while the pinned official repository snapshot exposes the 1,191-entry SOAR-RNA artifact but no corresponding SOAR-MultiOmics benchmark artifact."
      },
      "aliases": [
        "SOAR"
      ],
      "audit": {
        "audited_date": "2026-08-13",
        "notes": "Automated double-pass extraction plus deterministic official-resource pin; benchmark-wide artifact access is conflicted between the creator paper's availability statement and the pinned repository contents, the benchmark license remains unverified, and the benchmark date uses the version-of-record publication date as a visible provisional proxy.",
        "status": "audited-with-caveats",
        "unresolved_fields": 4
      },
      "benchmark_use_ids": [
        "single-cell-omics-arena-evaluation-of-large-language-m-single-cell-omics-arena-soar-1-use",
        "soar-e5d2b3e-rna-zero-shot-cot-use",
        "soar-e5d2b3e-rna-zero-shot-use"
      ],
      "capabilities": [
        "classification",
        "scientific-reasoning"
      ],
      "coverage_notes": [],
      "domains": [
        "transcriptomics",
        "single-cell",
        "multiomics",
        "epigenomics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "single-cell-omics-arena-soar-repository-result-snapshot"
      ],
      "evaluation_run_ids": [
        "soar-e5d2b3e-rna-zero-shot",
        "soar-e5d2b3e-rna-zero-shot-cot"
      ],
      "evidence": [
        {
          "accessed_date": "2026-08-13",
          "id": "single-cell-omics-arena-soar-automated-metadata-1-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "c70e5d9fcefa8afbb221eaef7d2e93705d1c6ca2ff364e3534246e7bc5016d4e",
            "type": "section",
            "value": "Data availability"
          },
          "source_id": "single-cell-omics-arena-evaluation-of-large-language-m",
          "source_type": "work",
          "supports": [
            "/access/artifacts"
          ]
        },
        {
          "accessed_date": "2026-08-13",
          "id": "single-cell-omics-arena-soar-automated-metadata-2-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "c70e5d9fcefa8afbb221eaef7d2e93705d1c6ca2ff364e3534246e7bc5016d4e",
            "type": "section",
            "value": "Data availability"
          },
          "source_id": "single-cell-omics-arena-evaluation-of-large-language-m",
          "source_type": "work",
          "supports": [
            "/access/level"
          ]
        },
        {
          "accessed_date": "2026-08-13",
          "id": "single-cell-omics-arena-soar-automated-metadata-3-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "c70e5d9fcefa8afbb221eaef7d2e93705d1c6ca2ff364e3534246e7bc5016d4e",
            "type": "section",
            "value": "Data availability"
          },
          "source_id": "single-cell-omics-arena-evaluation-of-large-language-m",
          "source_type": "work",
          "supports": [
            "/access/tasks"
          ]
        },
        {
          "accessed_date": "2026-08-13",
          "id": "single-cell-omics-arena-soar-automated-metadata-4-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "de21b7974f6bda3420b720bda1208d180185b8bc19507455f040de4b70e79e40",
            "type": "section",
            "value": "Introduction"
          },
          "source_id": "single-cell-omics-arena-evaluation-of-large-language-m",
          "source_type": "work",
          "supports": [
            "/aliases"
          ]
        },
        {
          "accessed_date": "2026-08-13",
          "id": "single-cell-omics-arena-soar-automated-metadata-5-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "d24f132443b80aa6b02b5758e0cf552bb2a8dbbf9581e1641a10dda6fa2fdb5a",
            "type": "section",
            "value": "Introduction"
          },
          "source_id": "single-cell-omics-arena-evaluation-of-large-language-m",
          "source_type": "work",
          "supports": [
            "/capabilities"
          ]
        },
        {
          "accessed_date": "2026-08-13",
          "id": "single-cell-omics-arena-soar-automated-metadata-6-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "a9e0633daae454df523eaf8ca61d45e201750686337ff29fa99435aeb5b46192",
            "type": "section",
            "value": "Evaluation benchmarks of SOAR-RNA and SOAR-MultiOmics"
          },
          "source_id": "single-cell-omics-arena-evaluation-of-large-language-m",
          "source_type": "work",
          "supports": [
            "/domains"
          ]
        },
        {
          "accessed_date": "2026-08-13",
          "id": "single-cell-omics-arena-soar-automated-metadata-7-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "7cd5783a825ad6b66111cbf8de662f25b3997421c7878bd390ed0a5306e05b9c",
            "type": "section",
            "value": "Evaluation benchmarks of SOAR-RNA and SOAR-MultiOmics"
          },
          "source_id": "single-cell-omics-arena-evaluation-of-large-language-m",
          "source_type": "work",
          "supports": [
            "/kind"
          ]
        },
        {
          "accessed_date": "2026-08-13",
          "id": "single-cell-omics-arena-soar-automated-metadata-8-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "9edddca741c29d3c52f571e255c2edda603df94b1b7894b02bb8e13ba3ea24e2",
            "type": "section",
            "value": "Method — prompting strategies and benchmark notation"
          },
          "source_id": "single-cell-omics-arena-evaluation-of-large-language-m",
          "source_type": "work",
          "supports": [
            "/modalities"
          ]
        },
        {
          "accessed_date": "2026-08-13",
          "id": "single-cell-omics-arena-soar-automated-metadata-9-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "c486f5ce117bbce4d6dfa323e955163bd7178cebec1e41f40fcf0e339cedd912",
            "type": "section",
            "value": "Introduction"
          },
          "source_id": "single-cell-omics-arena-evaluation-of-large-language-m",
          "source_type": "work",
          "supports": [
            "/name"
          ]
        },
        {
          "accessed_date": "2026-08-13",
          "id": "single-cell-omics-arena-soar-automated-metadata-10-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "b97505c0ba4a02b879186f245214349ca8ffe01cd57e8bb43b517eb749556cf6",
            "type": "other",
            "value": "Article metadata — author-affiliation mapping"
          },
          "source_id": "single-cell-omics-arena-evaluation-of-large-language-m",
          "source_type": "work",
          "supports": [
            "/organizations"
          ]
        },
        {
          "accessed_date": "2026-08-13",
          "id": "single-cell-omics-arena-soar-automated-metadata-11-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "7cda8fc140a516cc3601d22ee49001b2dc05869198fd33f439a596e5e4950f69",
            "type": "section",
            "value": "Evaluation benchmarks of SOAR-RNA and SOAR-MultiOmics"
          },
          "source_id": "single-cell-omics-arena-evaluation-of-large-language-m",
          "source_type": "work",
          "supports": [
            "/summary"
          ]
        },
        {
          "accessed_date": "2026-08-13",
          "id": "single-cell-omics-arena-soar-automated-bibliographic-evidence",
          "locator": {
            "document_page": null,
            "note": "Resolved from the canonical paper identifier during intake.",
            "printed_page": null,
            "source_fragment_sha256": "1353e596ded212738db1fd3419dcab03b23fd1b6639746e5271fad3a5132f49a",
            "type": "other",
            "value": "Crossref bibliographic metadata"
          },
          "source_id": "single-cell-omics-arena-evaluation-of-large-language-m",
          "source_type": "work",
          "supports": [
            "/release_date",
            "/versions/0/release_date"
          ]
        },
        {
          "accessed_date": "2026-08-13",
          "id": "single-cell-omics-arena-soar-automated-count-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "e8e501253b5bfebcfa04e9b28b573d0283cefcd8f26d0ccdb60eea112c341b4a",
            "type": "section",
            "value": "Introduction"
          },
          "source_id": "single-cell-omics-arena-evaluation-of-large-language-m",
          "source_type": "work",
          "supports": [
            "/task_counts/total",
            "/task_counts/basis",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis"
          ]
        },
        {
          "accessed_date": "2026-08-13",
          "id": "single-cell-omics-arena-soar-automated-subset-1-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "3b41c83ba886e786bd885abf2f26e9ecbd4f7962e4498d765b8823f670ab1dd8",
            "type": "section",
            "value": "SOAR-RNA — Cell-type normalization"
          },
          "source_id": "single-cell-omics-arena-evaluation-of-large-language-m",
          "source_type": "work",
          "supports": [
            "/task_counts/subsets",
            "/versions/0/task_counts/subsets",
            "/task_counts/subsets/0",
            "/versions/0/task_counts/subsets/0"
          ]
        },
        {
          "accessed_date": "2026-08-13",
          "id": "single-cell-omics-arena-soar-automated-resource-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "c70e5d9fcefa8afbb221eaef7d2e93705d1c6ca2ff364e3534246e7bc5016d4e",
            "type": "section",
            "value": "Data availability"
          },
          "source_id": "single-cell-omics-arena-evaluation-of-large-language-m",
          "source_type": "work",
          "supports": [
            "/resources",
            "/implementations",
            "/access/license"
          ]
        },
        {
          "accessed_date": "2026-08-13",
          "id": "single-cell-omics-arena-soar-pinned-repository-access-evidence",
          "locator": {
            "document_page": null,
            "note": "The pinned tree contains soar_benchmark/datasets/soar_rna.json and no SOAR-MultiOmics benchmark artifact.",
            "printed_page": null,
            "source_fragment_sha256": null,
            "type": "other",
            "value": "Repository tree at commit e5d2b3e2619cb56fece5fba78fae989a67fd0c13"
          },
          "source_id": "single-cell-omics-arena-soar-official-repository-resource",
          "source_type": "resource",
          "supports": [
            "/access/level",
            "/access/tasks",
            "/access/artifacts"
          ]
        },
        {
          "accessed_date": "2026-08-13",
          "id": "single-cell-omics-arena-soar-automated-creator-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "72c021e1c388147d0f2872db431ee6a818d0c750fbfc06e26be6ddd72667da7c",
            "type": "other",
            "value": "Article metadata and Introduction"
          },
          "source_id": "single-cell-omics-arena-evaluation-of-large-language-m",
          "source_type": "work",
          "supports": [
            "/resources/0"
          ]
        },
        {
          "accessed_date": "2026-08-13",
          "id": "single-cell-omics-arena-soar-automated-version-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "72c021e1c388147d0f2872db431ee6a818d0c750fbfc06e26be6ddd72667da7c",
            "type": "other",
            "value": "Article metadata and Introduction"
          },
          "source_id": "single-cell-omics-arena-evaluation-of-large-language-m",
          "source_type": "work",
          "supports": [
            "/latest_version",
            "/versions/0"
          ]
        },
        {
          "accessed_date": "2026-08-13",
          "id": "single-cell-omics-arena-soar-automated-task-1-evidence",
          "locator": {
            "document_page": null,
            "note": null,
            "printed_page": null,
            "source_fragment_sha256": "b50abcb504e9ac27c88463d3efc947f74ad82163fa201bff4d00f6259e74130c",
            "type": "section",
            "value": "Introduction"
          },
          "source_id": "single-cell-omics-arena-evaluation-of-large-language-m",
          "source_type": "work",
          "supports": [
            "/scientific_task_classification/entries/0"
          ]
        }
      ],
      "field_status": [
        {
          "confidence": "medium",
          "evidence_ids": [
            "single-cell-omics-arena-soar-automated-bibliographic-evidence"
          ],
          "path": "/release_date",
          "reason": "The creator source does not report a separate benchmark release date. The machine-readable value uses the version-of-record publication date as a proxy and must not be interpreted as an independently verified artifact release date.",
          "status": "provisional"
        },
        {
          "confidence": "medium",
          "evidence_ids": [
            "single-cell-omics-arena-soar-automated-bibliographic-evidence"
          ],
          "path": "/versions/0/release_date",
          "reason": "The initial-release snapshot is an Atlas label and its date uses the creator paper's version-of-record publication date because no formal benchmark version date is reported.",
          "status": "provisional"
        },
        {
          "confidence": "high",
          "evidence_ids": [
            "single-cell-omics-arena-soar-automated-metadata-2-evidence",
            "single-cell-omics-arena-soar-pinned-repository-access-evidence"
          ],
          "path": "/access/level",
          "reason": "The creator paper states that the benchmark datasets are publicly available, but the pinned official repository snapshot contains the SOAR-RNA artifact and no SOAR-MultiOmics benchmark artifact. The conservative Atlas value is partially-open pending a versioned MultiOmics release.",
          "status": "conflicted"
        },
        {
          "confidence": "high",
          "evidence_ids": [
            "single-cell-omics-arena-soar-automated-resource-evidence"
          ],
          "path": "/access/license",
          "reason": "The double-pass review verified the official resource identity but did not establish a redistributable benchmark license; the value remains null pending source-level license verification.",
          "status": "provisional"
        }
      ],
      "id": "single-cell-omics-arena-soar",
      "implementations": [
        {
          "commit": "e5d2b3e2619cb56fece5fba78fae989a67fd0c13",
          "framework": "official repository",
          "notes": "Commit resolved deterministically during paper intake.",
          "status": "official",
          "url": "https://github.com/aicb-ZhangLabs/SOAR"
        }
      ],
      "kind": "suite",
      "latest_version": "initial-release",
      "modalities": [
        "text",
        "raw-omics"
      ],
      "name": "Single-cell Omics Arena",
      "organizations": [
        "University of California, Irvine"
      ],
      "parent_id": null,
      "release_date": "2025-11-24",
      "resources": [
        {
          "access_notes": "Versioned creator source; full text is not mirrored.",
          "id": "single-cell-omics-arena-soar-creator-paper-resource",
          "last_checked": "2026-08-13",
          "license": null,
          "pin": null,
          "type": "paper",
          "url": "https://doi.org/10.1093/bib/bbaf622"
        },
        {
          "access_notes": "Official resource pinned immutably during intake.",
          "id": "single-cell-omics-arena-soar-official-repository-resource",
          "last_checked": "2026-08-13",
          "license": null,
          "pin": {
            "kind": "commit",
            "url": "https://github.com/aicb-ZhangLabs/SOAR/commit/e5d2b3e2619cb56fece5fba78fae989a67fd0c13",
            "value": "e5d2b3e2619cb56fece5fba78fae989a67fd0c13"
          },
          "type": "repository",
          "url": "https://github.com/aicb-ZhangLabs/SOAR"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "initial-release",
        "entries": [
          {
            "confidence": "high",
            "count": 1226,
            "count_basis": "Benchmark-wide cell-type annotation task total reported by the creator paper.",
            "count_ref": null,
            "count_unit": "tasks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "single-cell-omics-arena-soar-automated-task-1-evidence"
            ],
            "mapping_method": "official-taxonomy",
            "notes": null,
            "reporting_status": "reported",
            "task_type_id": "cell-type-annotation"
          }
        ],
        "notes": "Only high-confidence Scientific Tasks explicitly supported by the creator source are mapped.",
        "status": "partial"
      },
      "summary": "A benchmark for evaluating LLM cell-type annotation across scRNA-seq and single-cell multiomics data.",
      "task_counts": {
        "basis": "Benchmark-wide cell-type annotation task total reported in the abstract.",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "SOAR-RNA benchmark entries are reported by their annotated cell types after preprocessing.",
            "count": 1191,
            "exclusive": true,
            "exhaustive": false,
            "id": "soar-rna",
            "label": "SOAR-RNA cell-type annotation tasks after preprocessing",
            "notes": null,
            "partition_group": "soar-components",
            "reporting_status": "reported"
          }
        ],
        "total": 1226
      },
      "task_formats": [
        "unclassified"
      ],
      "verification": {
        "last_verified": "2026-08-13",
        "notes": "New family admitted after creator source and official resource verification; the provisional date, access conflict, and unresolved benchmark license are visibly flagged.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": null,
          "evidence_ids": [
            "single-cell-omics-arena-soar-automated-count-evidence",
            "single-cell-omics-arena-soar-automated-subset-1-evidence",
            "single-cell-omics-arena-soar-automated-version-evidence"
          ],
          "formal_tracks": [],
          "id": "single-cell-omics-arena-soar-initial-release-version",
          "label": "initial-release",
          "notes": "Atlas snapshot label for the creator-paper release; the source does not report a formal benchmark version.",
          "release_date": "2025-11-24",
          "status": "current",
          "task_counts": {
            "basis": "Benchmark-wide cell-type annotation task total reported in the abstract.",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "SOAR-RNA benchmark entries are reported by their annotated cell types after preprocessing.",
                "count": 1191,
                "exclusive": true,
                "exhaustive": false,
                "id": "soar-rna",
                "label": "SOAR-RNA cell-type annotation tasks after preprocessing",
                "notes": null,
                "partition_group": "soar-components",
                "reporting_status": "reported"
              }
            ],
            "total": 1226
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The repository releases example evaluations, representative trajectories, deterministic grader code, methods, and aggregate full-suite results.",
        "biosafety_notes": "The tasks analyze spatial biology datasets. This registry stores only metadata and result summaries and mirrors neither withheld data nor trajectories.",
        "grader": "Five deterministic grader families cover numeric tolerance, multiple choice, marker-gene precision/recall, label-set Jaccard, and distribution comparison.",
        "level": "partially-open",
        "license": "Apache-2.0 for the official repository; full withheld benchmark artifacts have no separately published redistribution grant",
        "tasks": "The full 159-evaluation set is withheld to reduce contamination; a representative sample covering every platform type and task category is public."
      },
      "aliases": [
        "Spatial Bench"
      ],
      "audit": {
        "audited_date": "2026-07-22",
        "notes": "Counts, access, graders, methods, and both version-specific result sets were checked against the creator paper and commit-pinned repository. Paper v2's two printed component partitions each sum to 147, conflicting with its stated 146 total.",
        "status": "audited-with-caveats",
        "unresolved_fields": 1
      },
      "benchmark_use_ids": [
        "anthropic-spatialbench-external-summary",
        "spatialbench-preprint-creation",
        "spatialbench-preprint-evaluation",
        "spatialbench-repository-creation",
        "spatialbench-repository-evaluation",
        "system-card-claude-opus-5-spatialbench-2-use"
      ],
      "capabilities": [
        "data-analysis",
        "coding",
        "tool-use",
        "scientific-reasoning"
      ],
      "coverage_notes": [
        {
          "count": 159,
          "coverage": "explicitly-in-scope",
          "notes": "Every current evaluation is drawn from a spatial transcriptomics workflow.",
          "reporting_status": "reported",
          "tag": "spatial-omics"
        },
        {
          "count": 159,
          "coverage": "explicitly-in-scope",
          "notes": "The official release describes all five platforms as spatial transcriptomics technologies.",
          "reporting_status": "reported",
          "tag": "transcriptomics"
        },
        {
          "count": null,
          "coverage": "observed",
          "notes": "Cell typing and clustering are explicit categories, but a standalone single-cell-only count is not published.",
          "reporting_status": "not_reported",
          "tag": "single-cell"
        }
      ],
      "domains": [
        "spatial-omics",
        "transcriptomics",
        "single-cell",
        "bioinformatics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "spatialbench-preprint",
        "spatialbench-repository-release",
        "system-card-claude-opus-5"
      ],
      "evaluation_run_ids": [
        "spatialbench-paper-v2-base",
        "spatialbench-paper-v2-claude-code",
        "spatialbench-paper-v2-latch",
        "spatialbench-repo-159-claude-code",
        "spatialbench-repo-159-mini-swe-agent",
        "spatialbench-repo-159-openai-codex",
        "spatialbench-repo-159-pi"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "spatialbench-evidence-identity",
          "locator": {
            "note": "Benchmark name, creators, objective, release, real-data setup, modalities, and deterministic evaluation design.",
            "type": "page",
            "value": "arXiv v2 pp. 1–3, abstract and §§1–2.1"
          },
          "source_id": "spatialbench-preprint",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/parent_id",
            "/organizations",
            "/release_date",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats"
          ]
        },
        {
          "accessed_date": "2026-07-22",
          "id": "spatialbench-evidence-paper-counts",
          "locator": {
            "note": "Complete 146-evaluation inventory and both category and platform partitions.",
            "type": "table",
            "value": "arXiv v2 Appendix A.1, Table 9 and Table 10"
          },
          "source_id": "spatialbench-preprint",
          "source_type": "work",
          "supports": [
            "/versions/0/release_date",
            "/versions/0/as_of",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets"
          ]
        },
        {
          "accessed_date": "2026-07-22",
          "id": "spatialbench-evidence-current-counts",
          "locator": {
            "note": "159 total evaluations; seven category counts and five platform counts.",
            "type": "repository-path",
            "value": "README.md, results/category_results.json, and results/platform_results.json at commit 5042c4f3ee597da1590650c7b894d068ae968e26"
          },
          "source_id": "spatialbench-repository-resource",
          "source_type": "resource",
          "supports": [
            "/latest_version",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/versions/1/release_date",
            "/versions/1/as_of",
            "/versions/1/task_counts/total",
            "/versions/1/task_counts/basis",
            "/versions/1/task_counts/subsets",
            "/coverage_notes",
            "/scientific_task_classification/entries/0",
            "/scientific_task_classification/entries/1",
            "/scientific_task_classification/entries/2",
            "/scientific_task_classification/entries/3"
          ]
        },
        {
          "accessed_date": "2026-07-22",
          "id": "spatialbench-evidence-access-grader",
          "locator": {
            "note": "Representative-only task release, Apache-2.0 code, deterministic grader families, runner, container methods, and public aggregate results.",
            "type": "repository-path",
            "value": "README.md, METHODS.md, LICENSE, example_evals/, results/, and spatialbench/ at commit 5042c4f3ee597da1590650c7b894d068ae968e26"
          },
          "source_id": "spatialbench-repository-resource",
          "source_type": "resource",
          "supports": [
            "/access/level",
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/access/biosafety_notes",
            "/resources",
            "/implementations"
          ]
        }
      ],
      "field_status": [
        {
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-evidence-paper-counts"
          ],
          "path": "/versions/0/task_counts/subsets",
          "reason": "Paper v2 Table 9 states 146 total evaluations, while both its seven category rows and five platform rows sum to 147; no correction is published.",
          "status": "conflicted"
        }
      ],
      "id": "spatialbench",
      "implementations": [
        {
          "commit": "5042c4f3ee597da1590650c7b894d068ae968e26",
          "framework": "SpatialBench with latch-eval-tools",
          "notes": "Public benchmark validation and runner code; the full evaluation set remains withheld.",
          "status": "official",
          "url": "https://github.com/latchbio/spatialbench/tree/5042c4f3ee597da1590650c7b894d068ae968e26"
        }
      ],
      "kind": "agentic-eval",
      "latest_version": "repo-159-5042c4f",
      "modalities": [
        "text",
        "table",
        "figure",
        "raw-omics",
        "image",
        "code"
      ],
      "name": "SpatialBench",
      "organizations": [
        "LatchBio"
      ],
      "parent_id": null,
      "release_date": "2025-12-26",
      "resources": [
        {
          "access_notes": "Version 2 creator preprint with the complete 146-evaluation inventory and printed result tables.",
          "id": "spatialbench-paper-resource",
          "last_checked": "2026-07-22",
          "license": null,
          "pin": {
            "kind": "snapshot",
            "url": "https://arxiv.org/pdf/2512.21907v2",
            "value": "sha256:2ffd1b523bf6291a97a0a98af1517d8cf6c8c30c51f13353dfe917de02a3b323"
          },
          "type": "paper",
          "url": "https://arxiv.org/pdf/2512.21907v2"
        },
        {
          "access_notes": "Official 159-evaluation benchmark, methods, sample tasks, trajectories, and result artifacts.",
          "id": "spatialbench-repository-resource",
          "last_checked": "2026-07-22",
          "license": "Apache-2.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/latchbio/spatialbench/tree/5042c4f3ee597da1590650c7b894d068ae968e26",
            "value": "5042c4f3ee597da1590650c7b894d068ae968e26"
          },
          "type": "repository",
          "url": "https://github.com/latchbio/spatialbench"
        },
        {
          "access_notes": "Deterministic grader definitions and benchmark runner at the same pinned snapshot.",
          "id": "spatialbench-grader-resource",
          "last_checked": "2026-07-22",
          "license": "Apache-2.0",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/latchbio/spatialbench/tree/5042c4f3ee597da1590650c7b894d068ae968e26/spatialbench",
            "value": "5042c4f3ee597da1590650c7b894d068ae968e26"
          },
          "type": "grader",
          "url": "https://github.com/latchbio/spatialbench/tree/5042c4f3ee597da1590650c7b894d068ae968e26/spatialbench"
        }
      ],
      "scientific_task_classification": {
        "as_of": "2026-06-10",
        "benchmark_version": "repo-159-5042c4f",
        "entries": [
          {
            "confidence": "high",
            "count": 45,
            "count_basis": "official category_results.json n_evals",
            "count_ref": "/task_counts/subsets/0/count",
            "count_unit": "problems",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "spatialbench-evidence-current-counts"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "Official Cell Typing category.",
            "reporting_status": "reported",
            "task_type_id": "cell-type-annotation"
          },
          {
            "confidence": "high",
            "count": 3,
            "count_basis": "official category_results.json n_evals",
            "count_ref": "/task_counts/subsets/1/count",
            "count_unit": "problems",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "spatialbench-evidence-current-counts"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "Official Clustering category.",
            "reporting_status": "reported",
            "task_type_id": "cell-state-clustering"
          },
          {
            "confidence": "high",
            "count": 40,
            "count_basis": "official category_results.json n_evals",
            "count_ref": "/task_counts/subsets/2/count",
            "count_unit": "problems",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "spatialbench-evidence-current-counts"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "Official Differential Expression category.",
            "reporting_status": "reported",
            "task_type_id": "differential-expression-analysis"
          },
          {
            "confidence": "high",
            "count": 36,
            "count_basis": "official category_results.json n_evals",
            "count_ref": "/task_counts/subsets/6/count",
            "count_unit": "problems",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "spatialbench-evidence-current-counts"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "Official Spatial Analysis category.",
            "reporting_status": "reported",
            "task_type_id": "spatial-omics-analysis"
          }
        ],
        "notes": "Four official categories map directly to existing leaf tasks. Dimensionality reduction, normalization, and QC remain visible as benchmark subsets without inventing new task terms in this intake.",
        "status": "partial"
      },
      "summary": "A benchmark of deterministic, verifiable agentic problems derived from real spatial-transcriptomics workflows, testing whether agents can manipulate data and recover key biological results.",
      "task_counts": {
        "basis": "evaluations in the official repository snapshot at commit 5042c4f3ee597da1590650c7b894d068ae968e26",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "official category_results.json n_evals",
            "count": 45,
            "exclusive": true,
            "exhaustive": true,
            "id": "spatialbench-159-cell-typing",
            "label": "Cell typing",
            "notes": "Category partition of the 159 evaluations.",
            "partition_group": "task-category",
            "reporting_status": "reported"
          },
          {
            "basis": "official category_results.json n_evals",
            "count": 3,
            "exclusive": true,
            "exhaustive": true,
            "id": "spatialbench-159-clustering",
            "label": "Clustering",
            "notes": "Category partition of the 159 evaluations.",
            "partition_group": "task-category",
            "reporting_status": "reported"
          },
          {
            "basis": "official category_results.json n_evals",
            "count": 40,
            "exclusive": true,
            "exhaustive": true,
            "id": "spatialbench-159-differential-expression",
            "label": "Differential expression",
            "notes": "Category partition of the 159 evaluations.",
            "partition_group": "task-category",
            "reporting_status": "reported"
          },
          {
            "basis": "official category_results.json n_evals",
            "count": 11,
            "exclusive": true,
            "exhaustive": true,
            "id": "spatialbench-159-dimensionality-reduction",
            "label": "Dimensionality reduction",
            "notes": "Category partition of the 159 evaluations.",
            "partition_group": "task-category",
            "reporting_status": "reported"
          },
          {
            "basis": "official category_results.json n_evals",
            "count": 7,
            "exclusive": true,
            "exhaustive": true,
            "id": "spatialbench-159-normalization",
            "label": "Normalization",
            "notes": "Category partition of the 159 evaluations.",
            "partition_group": "task-category",
            "reporting_status": "reported"
          },
          {
            "basis": "official category_results.json n_evals",
            "count": 17,
            "exclusive": true,
            "exhaustive": true,
            "id": "spatialbench-159-qc",
            "label": "Quality control",
            "notes": "Category partition of the 159 evaluations.",
            "partition_group": "task-category",
            "reporting_status": "reported"
          },
          {
            "basis": "official category_results.json n_evals",
            "count": 36,
            "exclusive": true,
            "exhaustive": true,
            "id": "spatialbench-159-spatial-analysis",
            "label": "Spatial analysis",
            "notes": "Category partition of the 159 evaluations.",
            "partition_group": "task-category",
            "reporting_status": "reported"
          },
          {
            "basis": "official platform_results.json n_evals",
            "count": 32,
            "exclusive": true,
            "exhaustive": true,
            "id": "spatialbench-159-atlasxomics",
            "label": "AtlasXOmics",
            "notes": "Platform partition of the same 159 evaluations; never added to category counts.",
            "partition_group": "platform",
            "reporting_status": "reported"
          },
          {
            "basis": "official platform_results.json n_evals",
            "count": 34,
            "exclusive": true,
            "exhaustive": true,
            "id": "spatialbench-159-curio",
            "label": "Curio",
            "notes": "Platform partition of the same 159 evaluations; never added to category counts.",
            "partition_group": "platform",
            "reporting_status": "reported"
          },
          {
            "basis": "official platform_results.json merfish n_evals",
            "count": 33,
            "exclusive": true,
            "exhaustive": true,
            "id": "spatialbench-159-merfish",
            "label": "MERFISH / Vizgen",
            "notes": "The result artifact uses merfish while the README names the vendor/platform as Vizgen.",
            "partition_group": "platform",
            "reporting_status": "reported"
          },
          {
            "basis": "official platform_results.json n_evals",
            "count": 30,
            "exclusive": true,
            "exhaustive": true,
            "id": "spatialbench-159-visium",
            "label": "Visium",
            "notes": "Platform partition of the same 159 evaluations; never added to category counts.",
            "partition_group": "platform",
            "reporting_status": "reported"
          },
          {
            "basis": "official platform_results.json n_evals",
            "count": 30,
            "exclusive": true,
            "exhaustive": true,
            "id": "spatialbench-159-xenium",
            "label": "Xenium",
            "notes": "Platform partition of the same 159 evaluations; never added to category counts.",
            "partition_group": "platform",
            "reporting_status": "reported"
          }
        ],
        "total": 159
      },
      "task_formats": [
        "containerized spatial-biology analysis problem",
        "deterministic graded agent episode"
      ],
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Paper v2 and the current repository snapshot are intentionally separated and never share a comparability group.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": "2026-01-05",
          "evidence_ids": [
            "spatialbench-evidence-paper-counts"
          ],
          "formal_tracks": [],
          "id": "spatialbench-paper-v2",
          "label": "paper-v2",
          "notes": "Complete paper-era suite used in the arXiv v2 model and harness experiments.",
          "release_date": "2026-01-05",
          "status": "superseded",
          "task_counts": {
            "basis": "problems in the complete arXiv v2 benchmark inventory",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "paper v2 Table 9",
                "count": 20,
                "exclusive": true,
                "exhaustive": false,
                "id": "spatialbench-v2-qc",
                "label": "Quality control",
                "notes": "Printed component count; the seven category rows sum to 147 rather than the stated 146 total.",
                "partition_group": "task-category",
                "reporting_status": "reported"
              },
              {
                "basis": "paper v2 Table 9",
                "count": 7,
                "exclusive": true,
                "exhaustive": false,
                "id": "spatialbench-v2-normalization",
                "label": "Normalization",
                "notes": "Printed component count; the seven category rows sum to 147 rather than the stated 146 total.",
                "partition_group": "task-category",
                "reporting_status": "reported"
              },
              {
                "basis": "paper v2 Table 9",
                "count": 15,
                "exclusive": true,
                "exhaustive": false,
                "id": "spatialbench-v2-dimensionality-reduction",
                "label": "Dimensionality reduction",
                "notes": "Printed component count; the seven category rows sum to 147 rather than the stated 146 total.",
                "partition_group": "task-category",
                "reporting_status": "reported"
              },
              {
                "basis": "paper v2 Table 9",
                "count": 22,
                "exclusive": true,
                "exhaustive": false,
                "id": "spatialbench-v2-clustering",
                "label": "Clustering",
                "notes": "Printed component count; the seven category rows sum to 147 rather than the stated 146 total.",
                "partition_group": "task-category",
                "reporting_status": "reported"
              },
              {
                "basis": "paper v2 Table 9",
                "count": 39,
                "exclusive": true,
                "exhaustive": false,
                "id": "spatialbench-v2-cell-typing",
                "label": "Cell typing",
                "notes": "Printed component count; the seven category rows sum to 147 rather than the stated 146 total.",
                "partition_group": "task-category",
                "reporting_status": "reported"
              },
              {
                "basis": "paper v2 Table 9",
                "count": 26,
                "exclusive": true,
                "exhaustive": false,
                "id": "spatialbench-v2-differential-expression",
                "label": "Differential expression",
                "notes": "Printed component count; the seven category rows sum to 147 rather than the stated 146 total.",
                "partition_group": "task-category",
                "reporting_status": "reported"
              },
              {
                "basis": "paper v2 Table 9",
                "count": 18,
                "exclusive": true,
                "exhaustive": false,
                "id": "spatialbench-v2-spatial-analysis",
                "label": "Spatial analysis",
                "notes": "Printed component count; the seven category rows sum to 147 rather than the stated 146 total.",
                "partition_group": "task-category",
                "reporting_status": "reported"
              },
              {
                "basis": "paper v2 Table 9",
                "count": 23,
                "exclusive": true,
                "exhaustive": false,
                "id": "spatialbench-v2-atlasxomics",
                "label": "AtlasXOmics",
                "notes": "Printed component count; the five platform rows sum to 147 rather than the stated 146 total.",
                "partition_group": "platform",
                "reporting_status": "reported"
              },
              {
                "basis": "paper v2 Table 9",
                "count": 19,
                "exclusive": true,
                "exhaustive": false,
                "id": "spatialbench-v2-merfish",
                "label": "MERFISH",
                "notes": "Printed component count; the five platform rows sum to 147 rather than the stated 146 total.",
                "partition_group": "platform",
                "reporting_status": "reported"
              },
              {
                "basis": "paper v2 Table 9",
                "count": 30,
                "exclusive": true,
                "exhaustive": false,
                "id": "spatialbench-v2-xenium",
                "label": "Xenium",
                "notes": "Printed component count; the five platform rows sum to 147 rather than the stated 146 total.",
                "partition_group": "platform",
                "reporting_status": "reported"
              },
              {
                "basis": "paper v2 Table 9",
                "count": 32,
                "exclusive": true,
                "exhaustive": false,
                "id": "spatialbench-v2-visium",
                "label": "Visium",
                "notes": "Printed component count; the five platform rows sum to 147 rather than the stated 146 total.",
                "partition_group": "platform",
                "reporting_status": "reported"
              },
              {
                "basis": "paper v2 Table 9",
                "count": 43,
                "exclusive": true,
                "exhaustive": false,
                "id": "spatialbench-v2-seeker",
                "label": "Seeker",
                "notes": "Printed component count; the five platform rows sum to 147 rather than the stated 146 total.",
                "partition_group": "platform",
                "reporting_status": "reported"
              }
            ],
            "total": 146
          }
        },
        {
          "as_of": "2026-06-10",
          "evidence_ids": [
            "spatialbench-evidence-current-counts"
          ],
          "formal_tracks": [],
          "id": "spatialbench-repo-159-5042c4f",
          "label": "repo-159-5042c4f",
          "notes": "Revised 159-evaluation snapshot; full tasks remain withheld, while results and a representative sample are public.",
          "release_date": "2026-06-10",
          "status": "current",
          "task_counts": {
            "basis": "evaluations in the official repository snapshot at commit 5042c4f3ee597da1590650c7b894d068ae968e26",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "official category_results.json n_evals",
                "count": 45,
                "exclusive": true,
                "exhaustive": true,
                "id": "spatialbench-159-cell-typing",
                "label": "Cell typing",
                "notes": "Category partition of the 159 evaluations.",
                "partition_group": "task-category",
                "reporting_status": "reported"
              },
              {
                "basis": "official category_results.json n_evals",
                "count": 3,
                "exclusive": true,
                "exhaustive": true,
                "id": "spatialbench-159-clustering",
                "label": "Clustering",
                "notes": "Category partition of the 159 evaluations.",
                "partition_group": "task-category",
                "reporting_status": "reported"
              },
              {
                "basis": "official category_results.json n_evals",
                "count": 40,
                "exclusive": true,
                "exhaustive": true,
                "id": "spatialbench-159-differential-expression",
                "label": "Differential expression",
                "notes": "Category partition of the 159 evaluations.",
                "partition_group": "task-category",
                "reporting_status": "reported"
              },
              {
                "basis": "official category_results.json n_evals",
                "count": 11,
                "exclusive": true,
                "exhaustive": true,
                "id": "spatialbench-159-dimensionality-reduction",
                "label": "Dimensionality reduction",
                "notes": "Category partition of the 159 evaluations.",
                "partition_group": "task-category",
                "reporting_status": "reported"
              },
              {
                "basis": "official category_results.json n_evals",
                "count": 7,
                "exclusive": true,
                "exhaustive": true,
                "id": "spatialbench-159-normalization",
                "label": "Normalization",
                "notes": "Category partition of the 159 evaluations.",
                "partition_group": "task-category",
                "reporting_status": "reported"
              },
              {
                "basis": "official category_results.json n_evals",
                "count": 17,
                "exclusive": true,
                "exhaustive": true,
                "id": "spatialbench-159-qc",
                "label": "Quality control",
                "notes": "Category partition of the 159 evaluations.",
                "partition_group": "task-category",
                "reporting_status": "reported"
              },
              {
                "basis": "official category_results.json n_evals",
                "count": 36,
                "exclusive": true,
                "exhaustive": true,
                "id": "spatialbench-159-spatial-analysis",
                "label": "Spatial analysis",
                "notes": "Category partition of the 159 evaluations.",
                "partition_group": "task-category",
                "reporting_status": "reported"
              },
              {
                "basis": "official platform_results.json n_evals",
                "count": 32,
                "exclusive": true,
                "exhaustive": true,
                "id": "spatialbench-159-atlasxomics",
                "label": "AtlasXOmics",
                "notes": "Platform partition of the same 159 evaluations; never added to category counts.",
                "partition_group": "platform",
                "reporting_status": "reported"
              },
              {
                "basis": "official platform_results.json n_evals",
                "count": 34,
                "exclusive": true,
                "exhaustive": true,
                "id": "spatialbench-159-curio",
                "label": "Curio",
                "notes": "Platform partition of the same 159 evaluations; never added to category counts.",
                "partition_group": "platform",
                "reporting_status": "reported"
              },
              {
                "basis": "official platform_results.json merfish n_evals",
                "count": 33,
                "exclusive": true,
                "exhaustive": true,
                "id": "spatialbench-159-merfish",
                "label": "MERFISH / Vizgen",
                "notes": "The result artifact uses merfish while the README names the vendor/platform as Vizgen.",
                "partition_group": "platform",
                "reporting_status": "reported"
              },
              {
                "basis": "official platform_results.json n_evals",
                "count": 30,
                "exclusive": true,
                "exhaustive": true,
                "id": "spatialbench-159-visium",
                "label": "Visium",
                "notes": "Platform partition of the same 159 evaluations; never added to category counts.",
                "partition_group": "platform",
                "reporting_status": "reported"
              },
              {
                "basis": "official platform_results.json n_evals",
                "count": 30,
                "exclusive": true,
                "exhaustive": true,
                "id": "spatialbench-159-xenium",
                "label": "Xenium",
                "notes": "Platform partition of the same 159 evaluations; never added to category counts.",
                "partition_group": "platform",
                "reporting_status": "reported"
              }
            ],
            "total": 159
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "The original TensorFlow implementation and a later PyTorch implementation publish loaders, training code, pretrained weights, and evaluators.",
        "biosafety_notes": "The registry stores no protein sequences or model outputs.",
        "grader": "Deterministic task-specific accuracy, precision, and rank-correlation scorers.",
        "level": "fully-open",
        "license": "MIT for the original implementation; benchmark datasets retain their source-specific terms and citation requirements.",
        "tasks": "The five supervised datasets, prescribed train/validation/test splits, and a Pfam pretraining corpus are publicly downloadable."
      },
      "aliases": [
        "Tasks Assessing Protein Embeddings"
      ],
      "audit": {
        "audited_date": "2026-07-22",
        "notes": "Task definitions and primary metrics were checked against the paper and original commit-pinned implementation.",
        "status": "audited",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "prediction",
        "classification",
        "regression"
      ],
      "coverage_notes": [
        {
          "count": 2,
          "coverage": "explicitly-in-scope",
          "notes": "Secondary-structure and contact-map prediction are distinct tasks; TAPE is not a full 3D folding benchmark.",
          "reporting_status": "reported",
          "tag": "protein-structure"
        },
        {
          "count": 2,
          "coverage": "explicitly-in-scope",
          "notes": "The paper calls fluorescence and stability protein-engineering tasks, but they evaluate prediction rather than sequence generation.",
          "reporting_status": "reported",
          "tag": "protein-design"
        }
      ],
      "domains": [
        "protein-sequence",
        "protein-structure",
        "protein-design"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "tape-paper"
      ],
      "evaluation_run_ids": [
        "tape-creator-full"
      ],
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "tape-paper-definition-evidence",
          "locator": {
            "note": "Defines the five tasks, their three biological groups, splits, primary metrics, and creator evaluation.",
            "type": "section",
            "value": "Abstract; Sections 4.2 and 5; pp. 4-7, Table 2"
          },
          "source_id": "tape-paper",
          "source_type": "work",
          "supports": [
            "/name",
            "/aliases",
            "/summary",
            "/kind",
            "/organizations",
            "/release_date",
            "/latest_version",
            "/domains",
            "/capabilities",
            "/modalities",
            "/task_formats",
            "/task_counts/total",
            "/task_counts/basis",
            "/task_counts/subsets",
            "/coverage_notes",
            "/access/level",
            "/access/license",
            "/resources",
            "/versions/0/release_date",
            "/versions/0/task_counts/total",
            "/versions/0/task_counts/basis",
            "/versions/0/task_counts/subsets",
            "/scientific_task_classification/entries/0",
            "/scientific_task_classification/entries/1",
            "/scientific_task_classification/entries/2",
            "/scientific_task_classification/entries/3",
            "/scientific_task_classification/entries/4"
          ]
        },
        {
          "accessed_date": "2026-07-22",
          "id": "tape-repository-evidence",
          "locator": {
            "note": "Confirms the five downloadable supervised datasets, evaluators, model list, and MIT code license.",
            "type": "repository-path",
            "value": "README.md; tape/tasks; tape/data_utils at dad242d0341379255213cb8715d33b57aa9369bb"
          },
          "source_id": "tape-original-repository-resource",
          "source_type": "resource",
          "supports": [
            "/access/tasks",
            "/access/artifacts",
            "/access/grader",
            "/access/license",
            "/resources",
            "/implementations",
            "/versions/0/formal_tracks"
          ]
        }
      ],
      "field_status": [],
      "id": "tape",
      "implementations": [
        {
          "commit": "dad242d0341379255213cb8715d33b57aa9369bb",
          "framework": "TAPE original TensorFlow implementation",
          "notes": "Paper-associated code, data download links, pretrained weights, and task evaluators.",
          "status": "official",
          "url": "https://github.com/songlab-cal/tape-neurips2019/tree/dad242d0341379255213cb8715d33b57aa9369bb"
        }
      ],
      "kind": "suite",
      "latest_version": "original-2019",
      "modalities": [
        "protein-sequence"
      ],
      "name": "TAPE",
      "organizations": [
        "University of California Berkeley"
      ],
      "parent_id": null,
      "release_date": "2019-06-19",
      "resources": [
        {
          "access_notes": "NeurIPS 2019 creator paper.",
          "id": "tape-paper-resource",
          "last_checked": "2026-07-22",
          "license": null,
          "pin": null,
          "type": "paper",
          "url": "https://proceedings.neurips.cc/paper/2019/file/37f65c068b7723cd7809ee2d31d7861c-Paper.pdf"
        },
        {
          "access_notes": "Original paper implementation; preferred over the later port for reproducing the paper tables.",
          "id": "tape-original-repository-resource",
          "last_checked": "2026-07-22",
          "license": "MIT",
          "pin": {
            "kind": "commit",
            "url": "https://github.com/songlab-cal/tape-neurips2019/tree/dad242d0341379255213cb8715d33b57aa9369bb",
            "value": "dad242d0341379255213cb8715d33b57aa9369bb"
          },
          "type": "repository",
          "url": "https://github.com/songlab-cal/tape-neurips2019"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "original-2019",
        "entries": [
          {
            "confidence": "high",
            "count": 1,
            "count_basis": "Supervised downstream benchmark tasks.",
            "count_ref": null,
            "count_unit": "tasks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "tape-paper-definition-evidence"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "Per-residue three-state and eight-state secondary-structure prediction constitute one benchmark task.",
            "reporting_status": "reported",
            "task_type_id": "protein-secondary-structure-prediction"
          },
          {
            "confidence": "high",
            "count": 1,
            "count_basis": "Supervised downstream benchmark tasks.",
            "count_ref": null,
            "count_unit": "tasks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "tape-paper-definition-evidence"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "Residue-pair contact prediction is the second structure task; it is not relabeled as full 3D folding.",
            "reporting_status": "reported",
            "task_type_id": "protein-contact-map-prediction"
          },
          {
            "confidence": "high",
            "count": 1,
            "count_basis": "Supervised downstream benchmark tasks.",
            "count_ref": null,
            "count_unit": "tasks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "tape-paper-definition-evidence"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "Fold-level remote-homology classification.",
            "reporting_status": "reported",
            "task_type_id": "protein-remote-homology-detection"
          },
          {
            "confidence": "high",
            "count": 1,
            "count_basis": "Supervised downstream benchmark tasks.",
            "count_ref": null,
            "count_unit": "tasks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "tape-paper-definition-evidence"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "Sequence-to-fluorescence regression; this is prediction, not sequence generation.",
            "reporting_status": "reported",
            "task_type_id": "protein-fluorescence-prediction"
          },
          {
            "confidence": "high",
            "count": 1,
            "count_basis": "Supervised downstream benchmark tasks.",
            "count_ref": null,
            "count_unit": "tasks",
            "coverage": "explicitly-in-scope",
            "evidence_ids": [
              "tape-paper-definition-evidence"
            ],
            "mapping_method": "official-taxonomy",
            "notes": "Sequence-to-stability regression; this is prediction, not sequence generation.",
            "reporting_status": "reported",
            "task_type_id": "protein-stability-prediction"
          }
        ],
        "notes": "The five supervised downstream tasks are explicitly defined and grouped in the creator paper.",
        "status": "complete"
      },
      "summary": "A five-task benchmark for protein representation learning spanning secondary structure, residue contacts, remote homology, fluorescence, and stability.",
      "task_counts": {
        "basis": "supervised downstream benchmark tasks",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "supervised downstream tasks",
            "count": 2,
            "exclusive": true,
            "exhaustive": true,
            "id": "tape-structure-tasks",
            "label": "Structure prediction tasks",
            "notes": "Secondary structure and contact prediction.",
            "reporting_status": "reported"
          },
          {
            "basis": "supervised downstream tasks",
            "count": 1,
            "exclusive": true,
            "exhaustive": true,
            "id": "tape-evolution-task",
            "label": "Evolutionary understanding task",
            "notes": "Remote homology detection.",
            "reporting_status": "reported"
          },
          {
            "basis": "supervised downstream tasks",
            "count": 2,
            "exclusive": true,
            "exhaustive": true,
            "id": "tape-engineering-tasks",
            "label": "Protein engineering tasks",
            "notes": "Fluorescence and stability landscape prediction.",
            "reporting_status": "reported"
          }
        ],
        "total": 5
      },
      "task_formats": [
        "sequence classification",
        "per-residue classification",
        "residue-pair classification",
        "sequence regression"
      ],
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Creator paper and original implementation agree on five supervised downstream tasks.",
        "status": "verified"
      },
      "versions": [
        {
          "as_of": null,
          "evidence_ids": [
            "tape-paper-definition-evidence",
            "tape-repository-evidence"
          ],
          "formal_tracks": [],
          "id": "tape-original-2019",
          "label": "original-2019",
          "notes": "Fixed snapshot of the five downstream tasks in the NeurIPS 2019 paper; language-model pretraining is not counted as a sixth downstream task.",
          "release_date": "2019-06-19",
          "status": "current",
          "task_counts": {
            "basis": "supervised downstream benchmark tasks",
            "reporting_status": "reported",
            "subsets": [
              {
                "basis": "supervised downstream tasks",
                "count": 2,
                "exclusive": true,
                "exhaustive": true,
                "id": "tape-structure-tasks",
                "label": "Structure prediction tasks",
                "notes": "Secondary structure and contact prediction.",
                "reporting_status": "reported"
              },
              {
                "basis": "supervised downstream tasks",
                "count": 1,
                "exclusive": true,
                "exhaustive": true,
                "id": "tape-evolution-task",
                "label": "Evolutionary understanding task",
                "notes": "Remote homology detection.",
                "reporting_status": "reported"
              },
              {
                "basis": "supervised downstream tasks",
                "count": 2,
                "exclusive": true,
                "exhaustive": true,
                "id": "tape-engineering-tasks",
                "label": "Protein engineering tasks",
                "notes": "Fluorescence and stability landscape prediction.",
                "reporting_status": "reported"
              }
            ],
            "total": 5
          }
        }
      ]
    },
    {
      "access": {
        "artifacts": "NCBI Virus is publicly accessible subject to its terms.",
        "biosafety_notes": "Sequence retrieval may involve pathogens; this registry does not reproduce sequences or queries.",
        "grader": "Manual verification is described; grader implementation is not public.",
        "level": "metadata-only",
        "license": null,
        "tasks": "Query construction and aggregate statistics are public; a runnable task package was not identified."
      },
      "aliases": [
        "Viral Sequence Benchmark"
      ],
      "audit": {
        "audited_date": null,
        "notes": "Not yet processed by the v1.1 field-level audit.",
        "status": "legacy",
        "unresolved_fields": 0
      },
      "benchmark_use_ids": [],
      "capabilities": [
        "retrieval",
        "tool-use",
        "data-analysis"
      ],
      "coverage_notes": [],
      "domains": [
        "virology",
        "genomics",
        "bioinformatics"
      ],
      "entity_type": "benchmark",
      "evaluating_work_ids": [
        "virbench-official"
      ],
      "evaluation_run_ids": [
        "virbench-official-run"
      ],
      "evidence": [
        {
          "locator": "VirBench methodology and results sections",
          "supports": [
            "summary",
            "task_counts",
            "domains",
            "capabilities",
            "modalities",
            "access"
          ],
          "work_id": "virbench-official"
        }
      ],
      "field_status": [],
      "id": "virbench",
      "implementations": [
        {
          "commit": null,
          "framework": "scientific agent with NCBI Virus tool",
          "notes": "Public description only.",
          "status": "not-available",
          "url": null
        }
      ],
      "kind": "agentic-eval",
      "latest_version": "1.0",
      "modalities": [
        "text",
        "dna-rna-sequence",
        "database",
        "web"
      ],
      "name": "VirBench",
      "organizations": [
        "Anthropic"
      ],
      "parent_id": null,
      "release_date": "2025-05-20",
      "resources": [
        {
          "access_notes": "Official Anthropic report.",
          "license": null,
          "type": "website",
          "url": "https://www.anthropic.com/research/agents-in-biology"
        }
      ],
      "scientific_task_classification": {
        "as_of": null,
        "benchmark_version": "1.0",
        "entries": [],
        "notes": "No evidence-backed scientific-task mapping is published for this record yet.",
        "status": "unclassified"
      },
      "summary": "Retrieval benchmark that tests whether scientific agents can answer verified viral-sequence questions by querying NCBI Virus.",
      "task_counts": {
        "basis": "manually verified queries",
        "reporting_status": "reported",
        "subsets": [
          {
            "basis": "pathogens",
            "count": 40,
            "exclusive": false,
            "exhaustive": false,
            "id": "pathogen-set",
            "label": "Pathogens represented",
            "notes": "Count is not a task partition.",
            "reporting_status": "reported"
          }
        ],
        "total": 120
      },
      "task_formats": [
        "database retrieval query"
      ],
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Aggregate design verified from the official research page.",
        "status": "verified"
      }
    }
  ],
  "changelog": [
    {
      "date": "2026-08-20",
      "entity_ids": [
        "explainable-protein-protein-binding-affinity-predictio",
        "not-reported-balm-ppi",
        "not-reported-balm-ppi-without-peft",
        "not-reported-balm-ppi-standard-regression-baseline",
        "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-1-use",
        "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-2-use",
        "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-3-use",
        "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-4-use",
        "explainable-protein-protein-binding-affinity-predictio-ab-bind-5-use",
        "explainable-protein-protein-binding-affinity-predictio-ab-bind-6-use",
        "explainable-protein-protein-binding-affinity-predictio-ab-bind-7-use",
        "explainable-protein-protein-binding-affinity-predictio-ab-bind-8-use",
        "explainable-protein-protein-binding-affinity-predictio-abbibench-9-use",
        "explainable-protein-protein-binding-affinity-predictio-abbibench-10-use",
        "explainable-protein-protein-binding-affinity-predictio-abbibench-11-use",
        "explainable-protein-protein-binding-affinity-predictio-abbibench-12-use"
      ],
      "summary": "AI-assisted double-pass paper intake (paper-intake:explainable-protein-protein-binding-affinity-predictio); production inclusion required owner approval of the final head SHA.",
      "type": "paper-intake",
      "version": "1.4.0-dev"
    },
    {
      "date": "2026-08-20",
      "entity_ids": [
        "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
        "abbibench",
        "not-reported-antiberty",
        "not-reported-antifold",
        "not-reported-currab",
        "not-reported-diffab",
        "not-reported-diffab-fixbb",
        "not-reported-dymean",
        "not-reported-dymean-fixbb",
        "not-reported-esm2",
        "not-reported-esm3",
        "not-reported-mean",
        "not-reported-mean-fixbb",
        "not-reported-progen2-large",
        "not-reported-prosst",
        "not-reported-saprot",
        "abbibench-a-benchmark-for-antibody-binding-affinity-ma-abbibench-1-use",
        "abbibench-a-benchmark-for-antibody-binding-affinity-ma-abbibench-2-use"
      ],
      "summary": "AI-assisted double-pass paper intake (paper-intake:abbibench-a-benchmark-for-antibody-binding-affinity-ma); production inclusion required owner approval of the final head SHA.",
      "type": "paper-intake",
      "version": "1.4.0-dev"
    },
    {
      "date": "2026-08-14",
      "entity_ids": [
        "ab-bind-antibody-binding-mutational-database-for-compu",
        "ab-bind",
        "accelrys-software-inc-discovery-studio",
        "not-reported-ddfire",
        "not-reported-dfire",
        "not-reported-foldx",
        "not-reported-rosetta",
        "not-reported-statium",
        "sirin-et-al-basa",
        "ab-bind-antibody-binding-mutational-database-for-compu-ab-bind-1-use",
        "ab-bind-antibody-binding-mutational-database-for-compu-ab-bind-2-use",
        "ab-bind-antibody-binding-mutational-database-for-compu-ab-bind-3-use"
      ],
      "summary": "AI-assisted double-pass paper intake (paper-intake:ab-bind-antibody-binding-mutational-database-for-compu); production inclusion required owner approval of the final head SHA.",
      "type": "paper-intake",
      "version": "1.4.0-dev"
    },
    {
      "date": "2026-08-13",
      "entity_ids": [
        "single-cell-omics-arena-soar-repository-result-snapshot",
        "openai-gpt-4o-2024-05-13",
        "openai-gpt-4o-mini-2024-07-18",
        "soar-e5d2b3e-rna-zero-shot-use",
        "soar-e5d2b3e-rna-zero-shot",
        "soar-e5d2b3e-rna-zero-shot-cot-use",
        "soar-e5d2b3e-rna-zero-shot-cot"
      ],
      "summary": "Commit-pinned SOAR official repository results normalized for the 1,191-entry SOAR-RNA formal subset with separate zero-shot and two-call chain-of-thought protocols.",
      "type": "paper-intake",
      "version": "1.4.0-dev"
    },
    {
      "date": "2026-08-13",
      "entity_ids": [
        "single-cell-omics-arena-evaluation-of-large-language-m",
        "single-cell-omics-arena-soar",
        "single-cell-omics-arena-evaluation-of-large-language-m-single-cell-omics-arena-soar-1-use"
      ],
      "summary": "AI-assisted double-pass paper intake (paper-intake:single-cell-omics-arena-evaluation-of-large-language-m); production inclusion required owner approval of the final head SHA.",
      "type": "paper-intake",
      "version": "1.4.0-dev"
    },
    {
      "date": "2026-08-11",
      "entity_ids": [
        "biosecbench-surveillance-repository-result-snapshot",
        "biosecbench-8d53fd8-pi-use",
        "biosecbench-8d53fd8-pi",
        "biosecbench-8d53fd8-claude-code-use",
        "biosecbench-8d53fd8-claude-code",
        "biosecbench-8d53fd8-openai-codex-use",
        "biosecbench-8d53fd8-openai-codex"
      ],
      "summary": "Commit-pinned BioSecBench-Surveillance official results normalized with an independent local Codex double-pass; the creator preprint and repository result snapshot remain separate source versions.",
      "type": "paper-intake",
      "version": "1.4.0-dev"
    },
    {
      "date": "2026-08-06",
      "entity_ids": [
        "comprehensive-benchmark-of-differential-transcript-usa",
        "comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu",
        "comprehensive-benchmark-of-differential-transcript-usa-comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-1-use",
        "comprehensive-benchmark-of-differential-transcript-usa-comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-2-use",
        "comprehensive-benchmark-of-differential-transcript-usa-comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-3-use",
        "comprehensive-benchmark-of-differential-transcript-usa-comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-4-use",
        "comprehensive-benchmark-of-differential-transcript-usa-comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-5-use",
        "comprehensive-benchmark-of-differential-transcript-usa-comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-6-use"
      ],
      "summary": "AI-assisted double-pass paper intake (paper-intake:comprehensive-benchmark-of-differential-transcript-usa); production inclusion required owner approval of the final head SHA.",
      "type": "paper-intake",
      "version": "1.4.0-dev"
    },
    {
      "date": "2026-08-02",
      "entity_ids": [
        "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
        "benchmark-human-crispr-cas9-library",
        "benchmark-dual-human-crispr-cas9-library",
        "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-benchmark-human-crispr-cas9-library-1-use",
        "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-benchmark-human-crispr-cas9-library-2-use",
        "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-benchmark-dual-human-crispr-cas9-library-3-use",
        "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-benchmark-dual-human-crispr-cas9-library-4-use"
      ],
      "summary": "AI-assisted double-pass paper intake (paper-intake:a-benchmark-comparison-of-crisprn-guide-rna-design-alg); production inclusion required owner approval of the final head SHA.",
      "type": "paper-intake",
      "version": "1.4.0-dev"
    },
    {
      "date": "2026-08-02",
      "entity_ids": [
        "enhancing-molecular-property-prediction-with-auxiliary",
        "hu-et-al-supervised-contextpred-sup-cp",
        "hu-et-al-supervised-sup",
        "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-1-use",
        "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-2-use",
        "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-3-use",
        "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-4-use"
      ],
      "summary": "AI-assisted double-pass paper intake (paper-intake:enhancing-molecular-property-prediction-with-auxiliary); production inclusion required owner approval of the final head SHA.",
      "type": "paper-intake",
      "version": "1.4.0-dev"
    },
    {
      "date": "2026-07-31",
      "entity_ids": [
        "an-integrative-approach-to-protein-sequence-design-thr",
        "rfah-benchmark",
        "papd-benchmark",
        "cam-benchmark",
        "jumper-et-al-alphafold2",
        "meier-et-al-esm-1v",
        "an-integrative-approach-to-protein-sequence-design-thr-rfah-benchmark-1-use",
        "an-integrative-approach-to-protein-sequence-design-thr-rfah-benchmark-2-use",
        "an-integrative-approach-to-protein-sequence-design-thr-papd-benchmark-3-use",
        "an-integrative-approach-to-protein-sequence-design-thr-papd-benchmark-4-use",
        "an-integrative-approach-to-protein-sequence-design-thr-cam-benchmark-5-use",
        "an-integrative-approach-to-protein-sequence-design-thr-cam-benchmark-6-use"
      ],
      "summary": "AI-assisted double-pass paper intake (paper-intake:an-integrative-approach-to-protein-sequence-design-thr); production inclusion required owner approval of the final head SHA.",
      "type": "paper-intake",
      "version": "1.4.0-dev"
    },
    {
      "date": "2026-07-31",
      "entity_ids": [
        "crafted-experiments-to-evaluate-feature-selection-meth",
        "crafted-experiments",
        "kim-zhou-and-chen-hippo",
        "liu-et-al-pp-anb",
        "liu-et-al-qq-anb",
        "liu-et-al-wdist-med",
        "satija-et-al-seurat-disp",
        "stuart-et-al-seurat-vst",
        "townes-et-al-deviancefs",
        "crafted-experiments-to-evaluate-feature-selection-meth-crafted-experiments-1-use",
        "crafted-experiments-to-evaluate-feature-selection-meth-crafted-experiments-2-use"
      ],
      "summary": "AI-assisted double-pass paper intake (paper-intake:crafted-experiments-to-evaluate-feature-selection-meth); production inclusion required owner approval of the final head SHA.",
      "type": "paper-intake",
      "version": "1.4.0-dev"
    },
    {
      "date": "2026-07-30",
      "entity_ids": [
        "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
        "biosecbench-surveillance",
        "anthropic-opus-4-6",
        "anthropic-opus-4-7",
        "anthropic-opus-4-8",
        "anthropic-sonnet-4-6",
        "xai-grok-4-20-reasoning",
        "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-1-use",
        "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-2-use",
        "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-3-use",
        "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-4-use"
      ],
      "summary": "AI-assisted double-pass paper intake (paper-intake:biosecbench-surveillance-a-verifiable-benchmark-for-ai); production inclusion required owner approval of the final head SHA.",
      "type": "paper-intake",
      "version": "1.4.0-dev"
    },
    {
      "date": "2026-07-29",
      "entity_ids": [
        "ppb-affinity-protein-protein-binding-affinity-dataset",
        "ppb-affinity",
        "liu-et-al-benchmark-algorithm",
        "ppb-affinity-protein-protein-binding-affinity-dataset-ppb-affinity-1-use",
        "ppb-affinity-protein-protein-binding-affinity-dataset-ppb-affinity-2-use",
        "ppb-affinity-protein-protein-binding-affinity-dataset-ppb-affinity-3-use"
      ],
      "summary": "AI-assisted double-pass paper intake (paper-intake:ppb-affinity-protein-protein-binding-affinity-dataset); production inclusion required owner approval of the final head SHA.",
      "type": "paper-intake",
      "version": "1.4.0-dev"
    },
    {
      "date": "2026-07-29",
      "entity_ids": [
        "system-card-claude-opus-5",
        "anthropic-claude-mythos-5",
        "anthropic-claude-opus-5",
        "anthropic-claude-sonnet-5",
        "system-card-claude-opus-5-biomysterybench-1-use",
        "system-card-claude-opus-5-spatialbench-2-use",
        "system-card-claude-opus-5-scbench-3-use",
        "system-card-claude-opus-5-proteingym-4-use"
      ],
      "summary": "AI-assisted double-pass paper intake (paper-intake:system-card-claude-opus-5); production inclusion required owner approval of the final head SHA.",
      "type": "paper-intake",
      "version": "1.4.0-dev"
    },
    {
      "date": "2026-07-27",
      "entity_ids": [
        "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an",
        "scbench",
        "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an-scbench-1-use",
        "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an-scbench-2-use",
        "scbench-gemini-2-5-pro-unversioned"
      ],
      "summary": "AI-assisted double-pass paper intake (paper-intake:scbench-evaluating-ai-agents-on-single-cell-rna-seq-an); production inclusion required owner approval of the final head SHA.",
      "type": "paper-intake",
      "version": "1.4.0-dev"
    },
    {
      "date": "2026-07-22",
      "entity_ids": [
        "registry-schema",
        "paper-discovery",
        "local-codex-double-pass",
        "paper-owner-gate"
      ],
      "summary": "Migrated guarded paper intake to an owner-triggered local Codex workflow while retaining weekly Europe PMC, Crossref, and arXiv candidate discovery. Two fresh read-only local sessions now perform extraction and independent verification; a local golden receipt gates production, deterministic code writes records, and an exact-head-SHA owner comment gates each paper PR. GitHub Actions no longer sends paper content to a model or requires an OpenAI API secret or GitHub App.",
      "type": "development",
      "version": "1.4.0-dev"
    },
    {
      "date": "2026-07-22",
      "entity_ids": [],
      "summary": "Released v1.3.1 as a dependency-only maintenance snapshot with grouped compatible Python, Astro, Node type, pnpm, and GitHub Pages action upgrades; retained TypeScript 5.9.3 after the TypeScript 7 update failed the project test suite.",
      "type": "release",
      "version": "1.3.1"
    },
    {
      "date": "2026-07-22",
      "entity_ids": [
        "registry-schema",
        "anthropic-life-sciences",
        "bixbench",
        "spatialbench",
        "spatialbench-preprint",
        "spatialbench-repository-release",
        "anthropic-healthcare-life-sciences",
        "anthropic-key-life-sciences-evals",
        "anthropic-scientific-figure-interpretation",
        "anthropic-computational-biology",
        "anthropic-protein-understanding",
        "claude-opus-4-5"
      ],
      "summary": "Released v1.3.0 with versioned Works and BenchmarkUse relationships; normalized Anthropic's partial BixBench claim, SpatialBench paper/current snapshots, an explicitly third-party SpatialBench summary, and three private internal life-science delta evaluations; added a draft-PR paper intake system and Paper Explorer without inferring missing settings or plot values.",
      "type": "release",
      "version": "1.3.0"
    },
    {
      "date": "2026-07-22",
      "entity_ids": [],
      "summary": "Released v1.2.0 with the Scientific Task Atlas taxonomy, evidence-backed benchmark mappings, normalized task-coverage exports, task explorer and detail pages, Chinese task-system guidance, and seven additional creator-audited benchmark families.",
      "type": "release",
      "version": "1.2.0"
    },
    {
      "date": "2026-07-22",
      "entity_ids": [
        "tape",
        "tape-paper",
        "tape-creator-full",
        "genomic-benchmarks",
        "genomic-benchmarks-paper",
        "genomic-benchmarks-creator-full",
        "beacon-rna",
        "beacon-paper",
        "beacon-creator-full",
        "moleculenet",
        "moleculenet-paper",
        "moleculenet-creator-full",
        "atom3d",
        "atom3d-paper",
        "atom3d-creator-full",
        "guacamol",
        "guacamol-paper",
        "guacamol-creator-full",
        "scib",
        "scib-paper",
        "scib-creator-full"
      ],
      "summary": "Added seven creator-audited benchmark families spanning protein representation learning, genomic sequence classification, RNA representation learning, molecular property prediction, 3D molecular learning, molecular generation, and single-cell integration; froze living implementations to commit-pinned snapshots and classified scientific tasks without flattening heterogeneous endpoints or metrics.",
      "type": "addition"
    },
    {
      "date": "2026-07-22",
      "entity_ids": [
        "registry-schema",
        "lifescibench",
        "proteingym",
        "casp",
        "cameo",
        "flip",
        "proteinlmbench",
        "bioinstruction",
        "lab-bench",
        "genebench-pro",
        "biomysterybench",
        "compbiobench",
        "bixbench",
        "blade",
        "scigym",
        "virbench"
      ],
      "summary": "Released v1.1.0 with field-level audits for fourteen launch families, structured versions and evidence, normalized official evaluation settings and results, and an explicit machine-validated VirBench legacy exception.",
      "type": "release"
    },
    {
      "date": "2026-07-21",
      "entity_ids": [
        "registry-schema",
        "virbench"
      ],
      "summary": "Closed the current audit pass with 14 of 15 launch families field-audited; retained VirBench as an intentionally deferred legacy record and added a machine-validated registry exception so no other unaudited record can enter production silently.",
      "type": "development"
    },
    {
      "date": "2026-07-21",
      "entity_ids": [
        "scigym",
        "scigym-small",
        "scigym-large",
        "scigym-paper",
        "scigym-small-creator-paper",
        "scigym-small-zero-shot",
        "scigym-gemini-2-5-flash-preview-04-17",
        "scigym-gemini-2-5-pro-preview-03-25",
        "scigym-gpt-4-1-2025-04-14",
        "scigym-gpt-4-1-mini-2025-04-14",
        "scigym-claude-3-5-haiku-20241022",
        "scigym-claude-3-7-sonnet-20250219"
      ],
      "summary": "Audited SCIGYM against the final NeurIPS paper, official project page, commit-pinned creator implementations, and immutable benchmark/evaluation Parquet snapshots; separated the 350 released SBML systems into 137 small and 213 large formal tracks, limited creator evaluation claims to the small track, registered all six exact model versions and 42 printed Table 1 values, reproduced three episodes per model-system pair, separated the no-tool zero-shot baseline, corrected the first public release date, and removed unsupported wet-lab, omics, protein/binding, and license claims.",
      "type": "audit"
    },
    {
      "date": "2026-07-21",
      "entity_ids": [
        "blade",
        "blade-mcq",
        "blade-analysis-generation",
        "blade-paper",
        "blade-creator-decision-mcq",
        "blade-creator-paper",
        "blade-creator-react",
        "blade-codellama-7b-instruct",
        "blade-deepseek-coder-6-7b-instruct",
        "blade-llama3-8b-unversioned",
        "blade-llama3-70b-unversioned",
        "blade-mixtral-8x22b-unversioned",
        "blade-gpt35-turbo-unversioned",
        "blade-gpt4o-unversioned",
        "blade-gemini-1-5-pro-unversioned",
        "blade-claude-3-5-sonnet-20240620"
      ],
      "summary": "Audited BLADE against the current creator manuscript, official project site, peer-reviewed paper, and commit-pinned package; separated 12 source research-question/dataset pairs from 188 MCQs and 536 ground-truth decision references, registered MCQ and end-to-end generation as formal tracks, corrected code/data licensing, limited biological coverage to four explicit source questions with no protein/binding/omics tasks, and split the creator evaluation into zero-shot MCQ, 40-repeat one-shot generation, and 20-repeat ten-step ReAct protocols with all 14 printed F1 results and confidence intervals.",
      "type": "audit"
    },
    {
      "date": "2026-07-21",
      "entity_ids": [
        "compbiobench",
        "compbiobench-preprint",
        "compbiobench-creator-full",
        "compbiobench-opus-full",
        "compbiobench-sonnet-full",
        "compbiobench-haiku-full",
        "compbiobench-codex-hardest",
        "compbiobench-opus-hardest",
        "compbiobench-sonnet-hardest",
        "compbiobench-haiku-hardest",
        "compbiobench-nonagentic-baselines",
        "codex-cli-gpt-5-4",
        "claude-code-opus-4-6",
        "claude-code-sonnet-4-6",
        "claude-code-haiku-4-5",
        "chatgpt-5-2"
      ],
      "summary": "Audited CompBioBench v1 against the immutable 100-row Zenodo TSV, public Hugging Face data mirror, official runner, leaderboard Space, and creator preprint; registered eight domain, five style, four difficulty, and internet-required partitions, corrected component licenses and access to partially open because answers/grading remain private, treated the single PDB projection task as protein structure without inferring binding coverage, and split all creator results by exact agent, effort, repeat, timeout, full/hardest scope, cost, time, and non-agentic protocol.",
      "type": "audit"
    },
    {
      "date": "2026-07-21",
      "entity_ids": [
        "biomysterybench",
        "biomysterybench-official",
        "biomysterybench-official-run",
        "biomysterybench-v8-human-solvable",
        "biomysterybench-v8-human-difficult",
        "claude-haiku-4-5",
        "claude-sonnet-4-6",
        "claude-opus-4-6",
        "claude-opus-4-7",
        "claude-mythos-preview"
      ],
      "summary": "Audited BioMysteryBench across versions; separated the current gated v11 release (90 problems, 73/17) from the superseded v8 creator evaluation (99 problems, 76/23), pinned the open preview and full-set commits plus three official result figures, corrected access and grader claims, and registered all ten printed subset scores across five models without inferring numeric confidence bounds.",
      "type": "audit"
    },
    {
      "date": "2026-07-21",
      "entity_ids": [
        "genebench-pro",
        "genebench-pro-report",
        "genebench-pro-official",
        "genebench-pro-pro-mode",
        "gpt-5-6-sol",
        "gpt-5-6-pro",
        "claude-opus-4-8"
      ],
      "summary": "Audited GeneBench-Pro as 129 synthetic problems partitioned into 10 public, 50 Artificial Analysis, and 69 internal-holdout problems; registered the 10-domain and 21-terminal-subdomain atlas plus 82/47 review strata, pinned the ten-problem public package and its reference grader, retained its CC-BY-4.0 versus MIT license conflict, and normalized all 60 creator-reported model configurations across 13 exact effort/repeat groups with confidence intervals, valid-attempt handling, tools, and binary-grader semantics.",
      "type": "audit"
    },
    {
      "date": "2026-07-21",
      "entity_ids": [
        "lab-bench",
        "lab-bench-paper",
        "lab-bench-litqa2",
        "lab-bench-suppqa",
        "lab-bench-figqa",
        "lab-bench-tableqa",
        "lab-bench-dbqa",
        "lab-bench-protocolqa",
        "lab-bench-seqqa",
        "lab-bench-cloning-scenarios",
        "anthropic-sonnet-4-5-system-card"
      ],
      "summary": "Audited LAB-Bench as 2,457 questions in eight broad categories; reproduced the 1,967 public and 490 private split across 31 versioned task files, preserved the README's conflicting 30-subtask claim, registered every formal DbQA and SeqQA child track, transcribed all 31 creator multiple-choice result rows and three expert-graded open-response studies, and separated Anthropic 10-shot and crop-tool evaluations by protocol and comparability group.",
      "type": "audit"
    },
    {
      "date": "2026-07-21",
      "entity_ids": [
        "bioinstruction",
        "bioinstructions-paper",
        "bioinstruction-emp",
        "bioinstruction-ea",
        "bioinstruction-pd300",
        "bioinstruction-cpd",
        "bioinstruction-tb-human",
        "bioinstruction-tb-mouse",
        "bioinstruction-apa",
        "bioinstruction-ncrna",
        "bioinstruction-modification",
        "bioinstruction-mrl",
        "bioinstruction-prs",
        "bioinstruction-crispr-on-target",
        "bioinstruction-ec",
        "bioinstruction-stability",
        "bioinstruction-fluorescence",
        "bioinstruction-solubility",
        "bioinstruction-thermostability",
        "bioinstruction-aan",
        "bioinstruction-rpi",
        "bioinstruction-epi",
        "bioinstruction-sirna"
      ],
      "summary": "Audited Biology-Instructions as 21 formal tasks rather than 21 questions; registered 6 DNA, 6 RNA, 5 protein, and 4 multi-molecule child tracks with train/validation/test counts, separated open-source, closed-source, and creator prompts into 63 runs, captured 345 creator-paper results, kept protein design out of scope, corrected access and license claims, and preserved the paper/repository conflicts over aggregate test totals, Stage-3 rows, and evaluator registration keys.",
      "type": "audit"
    },
    {
      "date": "2026-07-21",
      "entity_ids": [
        "proteinlmbench",
        "proteinlmbench-paper",
        "proteinlmbench-creator-full",
        "toursynbio-7b"
      ],
      "summary": "Audited ProteinLMBench against arXiv v2, its complete commit-pinned Hugging Face history and current JSON, and the official runner; corrected the release date to 2024-04-29, recorded the 2-to-10-option distribution and official license conflicts, kept topical design/binding counts unreported, and registered all 18 creator-evaluated systems with 36 accuracy and inference-time results.",
      "type": "audit"
    },
    {
      "date": "2026-07-21",
      "entity_ids": [
        "flip",
        "flip-aav",
        "flip-gb1",
        "flip-meltome",
        "flip-paper"
      ],
      "summary": "Audited original FLIP as 15 dataset-by-split tasks across three formal landscape tracks; separated task counts from sequence-example counts, registered all 15 creator evaluation protocols and 155 main-table Spearman results, isolated the two optimistic sampled splits, pinned the official implementation, and resolved the Meltome Human-cell total to 7,158 using the versioned CSV while retaining the paper's 7,156 discrepancy.",
      "type": "audit"
    },
    {
      "date": "2026-07-21",
      "entity_ids": [
        "cameo",
        "cameo-paper",
        "cameo-foundation-paper",
        "cameo-2024-ligand-baseline-common",
        "cameo-2024-ppi-three-server-common",
        "cameo-2024-antibody-three-server-common",
        "cameo-alphafold3-v301",
        "cameo-swissmodel-vina-vina",
        "cameo-swissmodel-vina-ad4",
        "cameo-swissmodel-glide",
        "cameo-multifold-2024",
        "cameo-swissmodel-2024"
      ],
      "summary": "Audited CAMEO as a weekly rolling service, corrected the creator-paper DOI and 2012 origin, separated the current complex-only category from discontinued legacy categories, preserved the bounded 7,150-target 2024 study and common subsets, and registered exact evaluated systems even where no scalar result is available.",
      "type": "audit"
    },
    {
      "date": "2026-07-21",
      "entity_ids": [
        "casp",
        "casp-protein-monomers",
        "casp-protein-multimers",
        "casp-protein-ligands",
        "casp-immune-complexes",
        "casp-official",
        "casp17-official",
        "casp16-monomer-assessment",
        "casp16-multimer-assessment",
        "casp16-ligand-assessment",
        "casp16-monomer-regular-official",
        "casp16-multimer-phase1-regular",
        "casp16-ligand-pose-regular",
        "casp16-ligand-affinity-stage1",
        "casp16-ligand-affinity-stage2"
      ],
      "summary": "Audited CASP as a round-based competition; separated the active CASP17 rolling snapshot from completed CASP16, registered monomer, multimer, ligand, and immune-complex tracks, preserved release/target/evaluation-unit count bases, and normalized five creator-assessor protocols without inventing a cross-category total or leaderboard.",
      "type": "audit"
    },
    {
      "date": "2026-07-21",
      "entity_ids": [
        "proteingym",
        "proteingym-dms-substitutions",
        "proteingym-dms-indels",
        "proteingym-clinical-substitutions",
        "proteingym-clinical-indels",
        "proteingym-paper",
        "proteingym-v10-dms-substitutions-zero-shot"
      ],
      "summary": "Audited ProteinGym v1.0-v1.3, separated assay/protein/variant count units into four formal tracks, recorded the Binding function category change from 14 to 13 DMS substitution assays, retained the official v1.3 indel count conflict (66 in the release archive versus 74 in the README), and normalized the 50-model v1.0 zero-shot DMS substitution evaluation.",
      "type": "audit"
    },
    {
      "date": "2026-07-21",
      "entity_ids": [
        "lifescibench",
        "lifescibench-preprint",
        "lifescibench-official-full"
      ],
      "summary": "Completed the LifeSciBench field-level audit; preserved 750/136/62, retained binding counts as Not reported, replaced an unsupported numbered version with an initial-release snapshot, and corrected unreported shot/repeat/grader settings.",
      "type": "audit"
    },
    {
      "date": "2026-07-21",
      "entity_ids": [
        "registry-schema"
      ],
      "summary": "Began the v1.1 field-level audit cycle with compatible versions, resource pins, structured evidence, audit status, provisional/conflicted claims, result confidence, and expanded source monitoring.",
      "type": "development"
    },
    {
      "date": "2026-07-21",
      "entity_ids": [
        "lifescibench",
        "genebench-pro",
        "biomysterybench",
        "virbench",
        "lab-bench",
        "bixbench",
        "scigym",
        "blade",
        "compbiobench",
        "proteingym",
        "casp",
        "cameo",
        "flip",
        "proteinlmbench",
        "bioinstruction"
      ],
      "summary": "Initial v1.0 registry with fifteen benchmark families, normalized official evaluations, schemas, exports, and the public atlas.",
      "type": "release"
    },
    {
      "date": "2026-07-21",
      "entity_ids": [
        "lifescibench"
      ],
      "summary": "Recorded 136 protein-primary-domain tasks, 62 protein design/optimization tasks, and explicit Not reported status for binding counts.",
      "type": "verification"
    },
    {
      "date": "2026-07-21",
      "entity_ids": [
        "biomysterybench"
      ],
      "summary": "Recorded the 99-question total, 76/23 human split, five-episode protocol, and accuracy/consistency metrics.",
      "type": "verification"
    },
    {
      "date": "2026-07-21",
      "entity_ids": [
        "bixbench",
        "bixbench-paper",
        "bixbench-v1-5-release",
        "bixbench-creator-paper",
        "bixbench-paper-mcq-refusal",
        "bixbench-paper-mcq-no-refusal",
        "bixbench-paper-mcq-no-images",
        "bixbench-v1-5-zero-shot-open",
        "bixbench-v1-5-zero-shot-mcq-refusal",
        "bixbench-v1-5-zero-shot-mcq-no-refusal",
        "bixbench-v1-5-agentic-open-images",
        "bixbench-v1-5-agentic-mcq-refusal-images",
        "bixbench-v1-5-agentic-mcq-no-refusal-images",
        "bixbench-v1-5-agentic-mcq-refusal-no-images",
        "bixbench-gpt-4o-unversioned",
        "bixbench-claude-3-5-sonnet-unversioned",
        "bixbench-claude-3-5-sonnet-20241022"
      ],
      "summary": "Audited BixBench across the original v1.0 paper snapshot and current v1.5 release; corrected the code/data license to Apache-2.0, separated 296 questions/53 capsules from 205 current questions, retained the unresolved 60-notebook claim beside 59 referenced capsule UUIDs and 64 stored archives, registered the 83/61/61 verifier partition and overlapping omics categories, and split creator agentic, ablation, and zero-shot protocols without digitizing unlabeled plots.",
      "type": "audit"
    }
  ],
  "evaluation_runs": [
    {
      "benchmark_id": "anthropic-computational-biology",
      "benchmark_version": "reported-2026-01-11",
      "comparability_group": "anthropic-computational-biology-delta",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "anthropic-computational-biology-delta-evidence",
          "locator": {
            "note": "Labels Accuracy (%) and annotates +10.5% from Opus 4.1 to Opus 4.5.",
            "type": "figure",
            "value": "Evals for key life sciences tasks — Computational biology"
          },
          "source_id": "anthropic-healthcare-life-sciences",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics",
            "/results"
          ]
        }
      ],
      "id": "anthropic-computational-biology-delta",
      "metrics": [
        {
          "aggregation": null,
          "baseline_model_id": "claude-opus-4-1",
          "higher_is_better": true,
          "kind": "delta",
          "metric_id": "accuracy-delta",
          "pass_threshold": null,
          "range": null,
          "source_label": "Accuracy improvement from Claude Opus 4.1 to Claude Opus 4.5",
          "tolerance": null,
          "unit": "percent delta as annotated"
        }
      ],
      "model_ids": [
        "claude-opus-4-1",
        "claude-opus-4-5",
        "claude-sonnet-4-5"
      ],
      "protocol": {
        "contamination": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": null,
          "model": null,
          "reporting_status": "not_reported",
          "type": null
        },
        "reasoning": {
          "notes": "Not reported for this benchmark direction.",
          "reporting_status": "not_reported",
          "value": null
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "statistical": {
          "notes": "No confidence interval or aggregation method is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "system_prompt_public": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "code_execution": {
            "notes": "Not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "container": {
            "notes": "Not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "external_tools": {
            "notes": "Not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "internet": {
            "notes": "Not reported.",
            "reporting_status": "not_reported",
            "value": null
          }
        },
        "turns": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "anthropic-computational-biology-delta-evidence"
          ],
          "metric_id": "accuracy-delta",
          "model_id": "claude-opus-4-5",
          "n": null,
          "notes": "Exact +10.5% chart annotation; no absolute accuracy or relative-versus-percentage-point interpretation is inferred.",
          "status": "verified",
          "value": 10.5
        }
      ],
      "scope": {
        "filter": null,
        "n": null,
        "reporting_status": "not_reported",
        "selection": null,
        "subset_id": null,
        "type": "unknown"
      },
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Only the labeled Opus 4.5 versus Opus 4.1 delta is normalized; plotted absolute positions are not digitized.",
        "status": "verified"
      },
      "work_id": "anthropic-healthcare-life-sciences",
      "work_version_id": "anthropic-healthcare-life-sciences-2026-01-11"
    },
    {
      "benchmark_id": "anthropic-protein-understanding",
      "benchmark_version": "reported-2026-01-11",
      "comparability_group": "anthropic-protein-understanding-delta",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "anthropic-protein-understanding-delta-evidence",
          "locator": {
            "note": "Labels Accuracy (%) and annotates +10.3% from Opus 4.1 to Opus 4.5.",
            "type": "figure",
            "value": "Evals for key life sciences tasks — Protein understanding"
          },
          "source_id": "anthropic-healthcare-life-sciences",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics",
            "/results"
          ]
        }
      ],
      "id": "anthropic-protein-understanding-delta",
      "metrics": [
        {
          "aggregation": null,
          "baseline_model_id": "claude-opus-4-1",
          "higher_is_better": true,
          "kind": "delta",
          "metric_id": "accuracy-delta",
          "pass_threshold": null,
          "range": null,
          "source_label": "Accuracy improvement from Claude Opus 4.1 to Claude Opus 4.5",
          "tolerance": null,
          "unit": "percent delta as annotated"
        }
      ],
      "model_ids": [
        "claude-opus-4-1",
        "claude-opus-4-5",
        "claude-sonnet-4-5"
      ],
      "protocol": {
        "contamination": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": null,
          "model": null,
          "reporting_status": "not_reported",
          "type": null
        },
        "reasoning": {
          "notes": "Not reported for this benchmark direction.",
          "reporting_status": "not_reported",
          "value": null
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "statistical": {
          "notes": "No confidence interval or aggregation method is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "system_prompt_public": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "code_execution": {
            "notes": "Not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "container": {
            "notes": "Not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "external_tools": {
            "notes": "Not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "internet": {
            "notes": "Not reported.",
            "reporting_status": "not_reported",
            "value": null
          }
        },
        "turns": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "anthropic-protein-understanding-delta-evidence"
          ],
          "metric_id": "accuracy-delta",
          "model_id": "claude-opus-4-5",
          "n": null,
          "notes": "Exact +10.3% chart annotation; no absolute accuracy or relative-versus-percentage-point interpretation is inferred.",
          "status": "verified",
          "value": 10.3
        }
      ],
      "scope": {
        "filter": null,
        "n": null,
        "reporting_status": "not_reported",
        "selection": null,
        "subset_id": null,
        "type": "unknown"
      },
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Only the labeled Opus 4.5 versus Opus 4.1 delta is normalized; plotted absolute positions are not digitized.",
        "status": "verified"
      },
      "work_id": "anthropic-healthcare-life-sciences",
      "work_version_id": "anthropic-healthcare-life-sciences-2026-01-11"
    },
    {
      "benchmark_id": "anthropic-scientific-figure-interpretation",
      "benchmark_version": "reported-2026-01-11",
      "comparability_group": "anthropic-scientific-figure-delta",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "anthropic-scientific-figure-delta-evidence",
          "locator": {
            "note": "Labels Accuracy (%) and annotates +13.2% from Opus 4.1 to Opus 4.5.",
            "type": "figure",
            "value": "Evals for key life sciences tasks — Scientific figure interpretation"
          },
          "source_id": "anthropic-healthcare-life-sciences",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics",
            "/results"
          ]
        }
      ],
      "id": "anthropic-scientific-figure-delta",
      "metrics": [
        {
          "aggregation": null,
          "baseline_model_id": "claude-opus-4-1",
          "higher_is_better": true,
          "kind": "delta",
          "metric_id": "accuracy-delta",
          "pass_threshold": null,
          "range": null,
          "source_label": "Accuracy improvement from Claude Opus 4.1 to Claude Opus 4.5",
          "tolerance": null,
          "unit": "percent delta as annotated"
        }
      ],
      "model_ids": [
        "claude-opus-4-1",
        "claude-opus-4-5",
        "claude-sonnet-4-5"
      ],
      "protocol": {
        "contamination": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": null,
          "model": null,
          "reporting_status": "not_reported",
          "type": null
        },
        "reasoning": {
          "notes": "Not reported for this benchmark direction.",
          "reporting_status": "not_reported",
          "value": null
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "statistical": {
          "notes": "No confidence interval or aggregation method is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "system_prompt_public": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "code_execution": {
            "notes": "Not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "container": {
            "notes": "Not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "external_tools": {
            "notes": "Not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "internet": {
            "notes": "Not reported.",
            "reporting_status": "not_reported",
            "value": null
          }
        },
        "turns": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "anthropic-scientific-figure-delta-evidence"
          ],
          "metric_id": "accuracy-delta",
          "model_id": "claude-opus-4-5",
          "n": null,
          "notes": "Exact +13.2% chart annotation; no absolute accuracy or relative-versus-percentage-point interpretation is inferred.",
          "status": "verified",
          "value": 13.2
        }
      ],
      "scope": {
        "filter": null,
        "n": null,
        "reporting_status": "not_reported",
        "selection": null,
        "subset_id": null,
        "type": "unknown"
      },
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Only the labeled Opus 4.5 versus Opus 4.1 delta is normalized; plotted absolute positions are not digitized.",
        "status": "verified"
      },
      "work_id": "anthropic-healthcare-life-sciences",
      "work_version_id": "anthropic-healthcare-life-sciences-2026-01-11"
    },
    {
      "benchmark_id": "atom3d",
      "benchmark_version": "v0.2.6",
      "comparability_group": "atom3d-creator-task-native",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "atom3d-creator-full-protocol-evidence",
          "locator": {
            "note": "Defines all eight datasets, official splits, model classes, native metrics, three-replicate aggregation, and creator baseline results.",
            "type": "table",
            "value": "Sections 2-5 and Tables 1-8; Appendix D"
          },
          "source_id": "atom3d-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/protocol",
            "/metrics"
          ]
        }
      ],
      "id": "atom3d-creator-full",
      "metrics": [
        {
          "aggregation": "held-out examples",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "mean-absolute-error",
          "pass_threshold": null,
          "range": null,
          "source_label": "MAE",
          "tolerance": null,
          "unit": "task-specific"
        },
        {
          "aggregation": "held-out examples",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "root-mean-squared-error",
          "pass_threshold": null,
          "range": null,
          "source_label": "RMSE",
          "tolerance": null,
          "unit": "task-specific"
        },
        {
          "aggregation": "held-out examples",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "auroc",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "AUROC",
          "tolerance": null,
          "unit": "area"
        },
        {
          "aggregation": "held-out examples",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "classification-accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Accuracy",
          "tolerance": null,
          "unit": "proportion"
        },
        {
          "aggregation": "task-specific",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "pearson-correlation",
          "pass_threshold": null,
          "range": [
            -1,
            1
          ],
          "source_label": "Pearson correlation",
          "tolerance": null,
          "unit": "correlation"
        },
        {
          "aggregation": "target-level structure ranking",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "spearman-correlation",
          "pass_threshold": null,
          "range": [
            -1,
            1
          ],
          "source_label": "Spearman correlation",
          "tolerance": null,
          "unit": "correlation"
        }
      ],
      "model_ids": [],
      "protocol": {
        "contamination": {
          "notes": "Users must preserve the selected official split.",
          "reporting_status": "reported",
          "value": "Task-specific sequence-identity, time, target, or scaffold splits."
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "deterministic task-specific scorer"
        },
        "reasoning": {
          "notes": "Supervised 3D molecular learning; no prompting.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "repeats": {
          "notes": "Table 8 aggregates creator baseline metrics and standard deviations over three replicates.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Exact seeds are not registered at suite level.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Supervised 3D molecular learning; no prompting.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "statistical": {
          "notes": "No cross-task total score.",
          "reporting_status": "reported",
          "value": "Native per-task metrics with standard deviations over three replicates; structure-ranking correlations are computed per target before summary."
        },
        "system_prompt_public": {
          "notes": "Supervised 3D molecular learning; no prompting.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "temperature": {
          "notes": "Supervised 3D molecular learning; no prompting.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "time_budget": {
          "notes": "No common task-wide runtime cap is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Supervised 3D molecular learning; no prompting.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Supervised 3D molecular learning; no prompting.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "code_execution": {
            "notes": "Training code is part of the benchmark procedure.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "container": {
            "notes": "Package dependencies are documented; no single evaluation container is prescribed.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Supervised 3D molecular learning; no prompting.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "external_tools": {
            "notes": "Not every architecture is applicable to every task.",
            "reporting_status": "reported",
            "value": "task-specific 3D CNN, graph neural network, and equivariant neural network pipelines"
          },
          "internet": {
            "notes": "Supervised 3D molecular learning; no prompting.",
            "reporting_status": "not_applicable",
            "value": null
          }
        },
        "turns": {
          "notes": "Supervised 3D molecular learning; no prompting.",
          "reporting_status": "not_applicable",
          "value": null
        }
      },
      "results": [],
      "scope": {
        "filter": "The eight ATOM3D task datasets evaluated in the creator paper.",
        "n": 8,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Eight-task scope, task-dependent architectures, splits, three-replicate aggregation, and metric families are verified.",
        "status": "verified"
      },
      "work_id": "atom3d-paper",
      "work_version_id": "atom3d-paper-2021-12-07"
    },
    {
      "benchmark_id": "beacon-rna",
      "benchmark_version": "neurips-2024",
      "comparability_group": "beacon-creator-task-native",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "beacon-creator-full-protocol-evidence",
          "locator": {
            "note": "Gives split sizes, task metrics, model families, three random seeds, and task-specific training settings.",
            "type": "table",
            "value": "Tables 1 and 3; Sections 5.1-5.2; Appendix A.1"
          },
          "source_id": "beacon-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/protocol",
            "/metrics"
          ]
        }
      ],
      "id": "beacon-creator-full",
      "metrics": [
        {
          "aggregation": "task-specific",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "f1-score",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "F1",
          "tolerance": null,
          "unit": "percent"
        },
        {
          "aggregation": "contact-map task",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision-at-l",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Top L Precision",
          "tolerance": null,
          "unit": "percent"
        },
        {
          "aggregation": "task-specific regression",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "r-squared",
          "pass_threshold": null,
          "range": null,
          "source_label": "R2",
          "tolerance": null,
          "unit": "percent"
        },
        {
          "aggregation": "splice-site task",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "top-k-accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Top-k ACC",
          "tolerance": null,
          "unit": "percent"
        },
        {
          "aggregation": "sequence classification",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "classification-accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "ACC",
          "tolerance": null,
          "unit": "percent"
        },
        {
          "aggregation": "modification prediction",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "auc",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "AUC",
          "tolerance": null,
          "unit": "percent"
        },
        {
          "aggregation": "mean columnwise RMSE",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "mcrmse",
          "pass_threshold": null,
          "range": null,
          "source_label": "MCRMSE",
          "tolerance": null,
          "unit": "error"
        },
        {
          "aggregation": "CRISPR task examples",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "spearman-correlation",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "Spearman Corr",
          "tolerance": null,
          "unit": "percent correlation"
        }
      ],
      "model_ids": [],
      "protocol": {
        "contamination": {
          "notes": "Table 1 provides every split size.",
          "reporting_status": "reported",
          "value": "Task-specific source datasets and prescribed train/validation/test partitions."
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "deterministic task-specific scorer"
        },
        "reasoning": {
          "notes": "Supervised fine-tuning benchmark; no in-context prompting.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "repeats": {
          "notes": "All experiments are repeated with three random seeds and tables report mean with variation.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "The identities of the three random seeds are not printed.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Supervised fine-tuning benchmark; no in-context prompting.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "statistical": {
          "notes": "Native metrics retain percent or error direction as printed.",
          "reporting_status": "reported",
          "value": "Task-level mean and variation over three seeds; no cross-task total score."
        },
        "system_prompt_public": {
          "notes": "Supervised fine-tuning benchmark; no in-context prompting.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "temperature": {
          "notes": "Supervised fine-tuning benchmark; no in-context prompting.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "time_budget": {
          "notes": "Task-specific epoch and batch settings are public; no common wall-clock cap is defined.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Supervised fine-tuning benchmark; no in-context prompting.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Supervised fine-tuning benchmark; no in-context prompting.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "code_execution": {
            "notes": "Model training is the evaluation procedure.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "container": {
            "notes": "Dependencies are pinned but no common container is supplied.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Supervised fine-tuning benchmark; no in-context prompting.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "external_tools": {
            "notes": "Tokenization and positional encoding are varied in separate ablations.",
            "reporting_status": "reported",
            "value": "task-specific fine-tuning scripts over CNN/ResNet/LSTM and RNA language models"
          },
          "internet": {
            "notes": "Supervised fine-tuning benchmark; no in-context prompting.",
            "reporting_status": "not_applicable",
            "value": null
          }
        },
        "turns": {
          "notes": "Supervised fine-tuning benchmark; no in-context prompting.",
          "reporting_status": "not_applicable",
          "value": null
        }
      },
      "results": [],
      "scope": {
        "filter": "All thirteen RNA tasks in final-paper Table 1.",
        "n": 13,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Full scope, all native metrics, and three-seed protocol are verified from final paper.",
        "status": "verified"
      },
      "work_id": "beacon-paper",
      "work_version_id": "beacon-paper-2024-12-10"
    },
    {
      "benchmark_id": "bioinstruction-aan",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-aan-closed-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-aan-closed-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (AAN test split); Appendix A.3; Table 8; Table 9 closed-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-aan-closed-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 7 (AAN column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-aan-closed-baselines",
      "metrics": [
        {
          "aggregation": "Matthews correlation coefficient across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "mcc",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "MCC",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-gpt4o",
        "bioinstruction-gpt4o-mini"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": "cardiffnlp/twitter-roberta-base-sentiment-latest fallback",
          "reporting_status": "reported",
          "type": "keyword-first binary extraction with sentiment-model fallback, followed by MCC"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests a direct JSON answer and no chain-of-thought."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes the complete biology-assistant system and user prompt for closed-source models.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-aan-closed-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-gpt4o-mini",
          "n": 3301,
          "notes": "AAN creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 1.59
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-aan-closed-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-gpt4o",
          "n": 3301,
          "notes": "AAN creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -3.29
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 3301,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-aan-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-aan",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-aan-creator-systems-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-aan-creator-systems-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (AAN test split); Appendix A.3; Table 8; Section 4.2 Psc prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-aan-creator-systems-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 7 (AAN column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-aan-creator-systems",
      "metrics": [
        {
          "aggregation": "Matthews correlation coefficient across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "mcc",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "MCC",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-chatmultiomics-balanced",
        "bioinstruction-chatmultiomics-stage12",
        "bioinstruction-chatmultiomics-stage123",
        "bioinstruction-chatmultiomics-stage2"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": "cardiffnlp/twitter-roberta-base-sentiment-latest fallback",
          "reporting_status": "reported",
          "type": "keyword-first binary extraction with sentiment-model fallback, followed by MCC"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "Psc requests clear, concise task answers and numeric output for regression tasks."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Section 4.2 publishes the Psc system prompt used for task performance computation.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-aan-creator-systems-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-chatmultiomics-balanced",
          "n": 3301,
          "notes": "AAN creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -1.48
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-aan-creator-systems-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-chatmultiomics-stage2",
          "n": 3301,
          "notes": "AAN creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.72
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-aan-creator-systems-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-chatmultiomics-stage12",
          "n": 3301,
          "notes": "AAN creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 10.26
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-aan-creator-systems-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-chatmultiomics-stage123",
          "n": 3301,
          "notes": "AAN creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 1.06
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 3301,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-aan-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-aan",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-aan-open-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-aan-open-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (AAN test split); Appendix A.3; Table 8; Table 9 open-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-aan-open-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 7 (AAN column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-aan-open-baselines",
      "metrics": [
        {
          "aggregation": "Matthews correlation coefficient across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "mcc",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "MCC",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-alpaca-7b",
        "bioinstruction-biomedgpt-lm-7b",
        "bioinstruction-galactica-13b",
        "bioinstruction-glm4-9b-chat",
        "bioinstruction-instructprotein-13b",
        "bioinstruction-llama-molinst-protein-7b",
        "bioinstruction-llama2-7b-chat",
        "bioinstruction-llama31-8b-instruct",
        "bioinstruction-qwen2-7b",
        "bioinstruction-vicuna15-7b"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": "cardiffnlp/twitter-roberta-base-sentiment-latest fallback",
          "reporting_status": "reported",
          "type": "keyword-first binary extraction with sentiment-model fallback, followed by MCC"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests the task-formatted answer and says not to explain or repeat."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes a user-prompt template and shows no system prompt for open-source models.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-aan-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-llama31-8b-instruct",
          "n": 3301,
          "notes": "AAN creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -1.05
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-aan-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-qwen2-7b",
          "n": 3301,
          "notes": "AAN creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 2.98
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-aan-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-llama2-7b-chat",
          "n": 3301,
          "notes": "AAN creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.63
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-aan-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-alpaca-7b",
          "n": 3301,
          "notes": "AAN creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.81
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-aan-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-glm4-9b-chat",
          "n": 3301,
          "notes": "AAN creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 1.32
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-aan-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-vicuna15-7b",
          "n": 3301,
          "notes": "AAN creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 2
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-aan-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-galactica-13b",
          "n": 3301,
          "notes": "AAN creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.01
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-aan-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-instructprotein-13b",
          "n": 3301,
          "notes": "AAN creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 1.53
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-aan-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-llama-molinst-protein-7b",
          "n": 3301,
          "notes": "AAN creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -1.38
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-aan-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-biomedgpt-lm-7b",
          "n": 3301,
          "notes": "AAN creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.92
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 3301,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-aan-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-apa",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-apa-closed-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-apa-closed-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (APA test split); Appendix A.3; Table 8; Table 9 closed-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-apa-closed-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 5 (APA column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-apa-closed-baselines",
      "metrics": [
        {
          "aggregation": "Squared Pearson correlation across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "r2",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "R2",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-gpt4o",
        "bioinstruction-gpt4o-mini"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "first-number extraction followed by squared Pearson correlation labeled R2"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests a direct JSON answer and no chain-of-thought."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes the complete biology-assistant system and user prompt for closed-source models.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-apa-closed-baselines-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-gpt4o-mini",
          "n": 49755,
          "notes": "APA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.05
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-apa-closed-baselines-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-gpt4o",
          "n": 49755,
          "notes": "APA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 49755,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-apa-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-apa",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-apa-creator-systems-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-apa-creator-systems-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (APA test split); Appendix A.3; Table 8; Section 4.2 Psc prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-apa-creator-systems-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 5 (APA column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-apa-creator-systems",
      "metrics": [
        {
          "aggregation": "Squared Pearson correlation across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "r2",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "R2",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-chatmultiomics-balanced",
        "bioinstruction-chatmultiomics-stage12",
        "bioinstruction-chatmultiomics-stage123",
        "bioinstruction-chatmultiomics-stage2"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "first-number extraction followed by squared Pearson correlation labeled R2"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "Psc requests clear, concise task answers and numeric output for regression tasks."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Section 4.2 publishes the Psc system prompt used for task performance computation.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-apa-creator-systems-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-chatmultiomics-balanced",
          "n": 49755,
          "notes": "APA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.01
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-apa-creator-systems-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-chatmultiomics-stage2",
          "n": 49755,
          "notes": "APA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-apa-creator-systems-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-chatmultiomics-stage12",
          "n": 49755,
          "notes": "APA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 50.68
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-apa-creator-systems-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-chatmultiomics-stage123",
          "n": 49755,
          "notes": "APA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 59.01
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 49755,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-apa-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-apa",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-apa-open-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-apa-open-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (APA test split); Appendix A.3; Table 8; Table 9 open-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-apa-open-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 5 (APA column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-apa-open-baselines",
      "metrics": [
        {
          "aggregation": "Squared Pearson correlation across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "r2",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "R2",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-alpaca-7b",
        "bioinstruction-biomedgpt-lm-7b",
        "bioinstruction-galactica-13b",
        "bioinstruction-glm4-9b-chat",
        "bioinstruction-instructprotein-13b",
        "bioinstruction-llama-molinst-protein-7b",
        "bioinstruction-llama2-7b-chat",
        "bioinstruction-llama31-8b-instruct",
        "bioinstruction-qwen2-7b",
        "bioinstruction-vicuna15-7b"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "first-number extraction followed by squared Pearson correlation labeled R2"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests the task-formatted answer and says not to explain or repeat."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes a user-prompt template and shows no system prompt for open-source models.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-apa-open-baselines-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-llama31-8b-instruct",
          "n": 49755,
          "notes": "APA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.01
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-apa-open-baselines-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-qwen2-7b",
          "n": 49755,
          "notes": "APA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-apa-open-baselines-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-llama2-7b-chat",
          "n": 49755,
          "notes": "APA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-apa-open-baselines-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-alpaca-7b",
          "n": 49755,
          "notes": "APA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-apa-open-baselines-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-glm4-9b-chat",
          "n": 49755,
          "notes": "APA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-apa-open-baselines-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-vicuna15-7b",
          "n": 49755,
          "notes": "APA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.01
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-apa-open-baselines-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-galactica-13b",
          "n": 49755,
          "notes": "APA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-apa-open-baselines-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-instructprotein-13b",
          "n": 49755,
          "notes": "APA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-apa-open-baselines-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-llama-molinst-protein-7b",
          "n": 49755,
          "notes": "APA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.02
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-apa-open-baselines-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-biomedgpt-lm-7b",
          "n": 49755,
          "notes": "APA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 49755,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-apa-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-cpd",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-cpd-closed-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-cpd-closed-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (CPD test split); Appendix A.3; Table 8; Table 9 closed-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-cpd-closed-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 4 (CPD column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-cpd-closed-baselines",
      "metrics": [
        {
          "aggregation": "Matthews correlation coefficient across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "mcc",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "MCC",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-gpt4o",
        "bioinstruction-gpt4o-mini"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": "cardiffnlp/twitter-roberta-base-sentiment-latest fallback",
          "reporting_status": "reported",
          "type": "keyword-first binary extraction with sentiment-model fallback, followed by MCC"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests a direct JSON answer and no chain-of-thought."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes the complete biology-assistant system and user prompt for closed-source models.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-cpd-closed-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-gpt4o-mini",
          "n": 11840,
          "notes": "CPD creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -2.95
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-cpd-closed-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-gpt4o",
          "n": 11840,
          "notes": "CPD creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.84
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 11840,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-cpd-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-cpd",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-cpd-creator-systems-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-cpd-creator-systems-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (CPD test split); Appendix A.3; Table 8; Section 4.2 Psc prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-cpd-creator-systems-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 4 (CPD column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-cpd-creator-systems",
      "metrics": [
        {
          "aggregation": "Matthews correlation coefficient across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "mcc",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "MCC",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-chatmultiomics-balanced",
        "bioinstruction-chatmultiomics-stage12",
        "bioinstruction-chatmultiomics-stage123",
        "bioinstruction-chatmultiomics-stage2"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": "cardiffnlp/twitter-roberta-base-sentiment-latest fallback",
          "reporting_status": "reported",
          "type": "keyword-first binary extraction with sentiment-model fallback, followed by MCC"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "Psc requests clear, concise task answers and numeric output for regression tasks."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Section 4.2 publishes the Psc system prompt used for task performance computation.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-cpd-creator-systems-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-chatmultiomics-balanced",
          "n": 11840,
          "notes": "CPD creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 5.57
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-cpd-creator-systems-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-chatmultiomics-stage2",
          "n": 11840,
          "notes": "CPD creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 1.8
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-cpd-creator-systems-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-chatmultiomics-stage12",
          "n": 11840,
          "notes": "CPD creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 41.18
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-cpd-creator-systems-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-chatmultiomics-stage123",
          "n": 11840,
          "notes": "CPD creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 44.54
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 11840,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-cpd-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-cpd",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-cpd-open-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-cpd-open-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (CPD test split); Appendix A.3; Table 8; Table 9 open-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-cpd-open-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 4 (CPD column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-cpd-open-baselines",
      "metrics": [
        {
          "aggregation": "Matthews correlation coefficient across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "mcc",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "MCC",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-alpaca-7b",
        "bioinstruction-galactica-13b",
        "bioinstruction-glm4-9b-chat",
        "bioinstruction-instructprotein-13b",
        "bioinstruction-llama-molinst-protein-7b",
        "bioinstruction-llama2-7b-chat",
        "bioinstruction-llama31-8b-instruct",
        "bioinstruction-qwen2-7b",
        "bioinstruction-vicuna15-7b"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": "cardiffnlp/twitter-roberta-base-sentiment-latest fallback",
          "reporting_status": "reported",
          "type": "keyword-first binary extraction with sentiment-model fallback, followed by MCC"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests the task-formatted answer and says not to explain or repeat."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes a user-prompt template and shows no system prompt for open-source models.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-cpd-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-llama31-8b-instruct",
          "n": 11840,
          "notes": "CPD creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-cpd-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-qwen2-7b",
          "n": 11840,
          "notes": "CPD creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 1.35
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-cpd-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-llama2-7b-chat",
          "n": 11840,
          "notes": "CPD creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.55
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-cpd-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-alpaca-7b",
          "n": 11840,
          "notes": "CPD creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -1.3
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-cpd-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-glm4-9b-chat",
          "n": 11840,
          "notes": "CPD creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -2.53
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-cpd-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-vicuna15-7b",
          "n": 11840,
          "notes": "CPD creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-cpd-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-galactica-13b",
          "n": 11840,
          "notes": "CPD creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -1.01
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-cpd-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-instructprotein-13b",
          "n": 11840,
          "notes": "CPD creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.33
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-cpd-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-llama-molinst-protein-7b",
          "n": 11840,
          "notes": "CPD creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 1.98
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 11840,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-cpd-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-crispr-on-target",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-crispr-on-target-closed-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-crispr-on-target-closed-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (CRI-On test split); Appendix A.3; Table 8; Table 9 closed-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-crispr-on-target-closed-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 5 (CRI-On column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-crispr-on-target-closed-baselines",
      "metrics": [
        {
          "aggregation": "Spearman rank correlation across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "spearman-rho",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "Spearman's ρ",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-gpt4o",
        "bioinstruction-gpt4o-mini"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "first-number extraction followed by Spearman rank correlation"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests a direct JSON answer and no chain-of-thought."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes the complete biology-assistant system and user prompt for closed-source models.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-crispr-on-target-closed-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-gpt4o-mini",
          "n": 416,
          "notes": "CRI-On creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 3.77
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-crispr-on-target-closed-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-gpt4o",
          "n": 416,
          "notes": "CRI-On creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -3.31
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 416,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-crispr-on-target-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-crispr-on-target",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-crispr-on-target-creator-systems-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-crispr-on-target-creator-systems-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (CRI-On test split); Appendix A.3; Table 8; Section 4.2 Psc prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-crispr-on-target-creator-systems-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 5 (CRI-On column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-crispr-on-target-creator-systems",
      "metrics": [
        {
          "aggregation": "Spearman rank correlation across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "spearman-rho",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "Spearman's ρ",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-chatmultiomics-balanced",
        "bioinstruction-chatmultiomics-stage12",
        "bioinstruction-chatmultiomics-stage123",
        "bioinstruction-chatmultiomics-stage2"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "first-number extraction followed by Spearman rank correlation"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "Psc requests clear, concise task answers and numeric output for regression tasks."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Section 4.2 publishes the Psc system prompt used for task performance computation.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-crispr-on-target-creator-systems-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-chatmultiomics-balanced",
          "n": 416,
          "notes": "CRI-On creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.31
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-crispr-on-target-creator-systems-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-chatmultiomics-stage2",
          "n": 416,
          "notes": "CRI-On creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 2.87
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-crispr-on-target-creator-systems-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-chatmultiomics-stage12",
          "n": 416,
          "notes": "CRI-On creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -2.99
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-crispr-on-target-creator-systems-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-chatmultiomics-stage123",
          "n": 416,
          "notes": "CRI-On creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.02
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 416,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-crispr-on-target-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-crispr-on-target",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-crispr-on-target-open-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-crispr-on-target-open-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (CRI-On test split); Appendix A.3; Table 8; Table 9 open-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-crispr-on-target-open-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 5 (CRI-On column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-crispr-on-target-open-baselines",
      "metrics": [
        {
          "aggregation": "Spearman rank correlation across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "spearman-rho",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "Spearman's ρ",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-alpaca-7b",
        "bioinstruction-biomedgpt-lm-7b",
        "bioinstruction-galactica-13b",
        "bioinstruction-glm4-9b-chat",
        "bioinstruction-instructprotein-13b",
        "bioinstruction-llama-molinst-protein-7b",
        "bioinstruction-llama2-7b-chat",
        "bioinstruction-llama31-8b-instruct",
        "bioinstruction-qwen2-7b",
        "bioinstruction-vicuna15-7b"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "first-number extraction followed by Spearman rank correlation"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests the task-formatted answer and says not to explain or repeat."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes a user-prompt template and shows no system prompt for open-source models.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-crispr-on-target-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-llama31-8b-instruct",
          "n": 416,
          "notes": "CRI-On creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.09
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-crispr-on-target-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-qwen2-7b",
          "n": 416,
          "notes": "CRI-On creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -6.21
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-crispr-on-target-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-llama2-7b-chat",
          "n": 416,
          "notes": "CRI-On creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.92
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-crispr-on-target-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-alpaca-7b",
          "n": 416,
          "notes": "CRI-On creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -3.55
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-crispr-on-target-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-glm4-9b-chat",
          "n": 416,
          "notes": "CRI-On creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.02
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-crispr-on-target-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-vicuna15-7b",
          "n": 416,
          "notes": "CRI-On creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 1.88
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-crispr-on-target-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-galactica-13b",
          "n": 416,
          "notes": "CRI-On creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -5.56
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-crispr-on-target-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-instructprotein-13b",
          "n": 416,
          "notes": "CRI-On creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-crispr-on-target-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-llama-molinst-protein-7b",
          "n": 416,
          "notes": "CRI-On creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.1
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-crispr-on-target-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-biomedgpt-lm-7b",
          "n": 416,
          "notes": "CRI-On creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.12
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 416,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-crispr-on-target-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-ea",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-ea-closed-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-ea-closed-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (EA test split); Appendix A.3; Table 8; Table 9 closed-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-ea-closed-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 4 (EA column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-ea-closed-baselines",
      "metrics": [
        {
          "aggregation": "Pearson correlation across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "pcc-housekeeping",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "PCC (housekeeping enhancer activity)",
          "tolerance": null,
          "unit": "percent"
        },
        {
          "aggregation": "Pearson correlation across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "pcc-developmental",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "PCC (developmental enhancer activity)",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-gpt4o",
        "bioinstruction-gpt4o-mini"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "two-number extraction and separate Pearson correlations"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests a direct JSON answer and no chain-of-thought."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes the complete biology-assistant system and user prompt for closed-source models.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ea-closed-baselines-results"
          ],
          "metric_id": "pcc-housekeeping",
          "model_id": "bioinstruction-gpt4o-mini",
          "n": 41186,
          "notes": "EA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.76
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ea-closed-baselines-results"
          ],
          "metric_id": "pcc-developmental",
          "model_id": "bioinstruction-gpt4o-mini",
          "n": 41186,
          "notes": "EA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.09
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ea-closed-baselines-results"
          ],
          "metric_id": "pcc-housekeeping",
          "model_id": "bioinstruction-gpt4o",
          "n": 41186,
          "notes": "EA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -1.17
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ea-closed-baselines-results"
          ],
          "metric_id": "pcc-developmental",
          "model_id": "bioinstruction-gpt4o",
          "n": 41186,
          "notes": "EA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -1.49
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 41186,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-ea-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-ea",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-ea-creator-systems-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-ea-creator-systems-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (EA test split); Appendix A.3; Table 8; Section 4.2 Psc prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-ea-creator-systems-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 4 (EA column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-ea-creator-systems",
      "metrics": [
        {
          "aggregation": "Pearson correlation across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "pcc-housekeeping",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "PCC (housekeeping enhancer activity)",
          "tolerance": null,
          "unit": "percent"
        },
        {
          "aggregation": "Pearson correlation across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "pcc-developmental",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "PCC (developmental enhancer activity)",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-chatmultiomics-balanced",
        "bioinstruction-chatmultiomics-stage12",
        "bioinstruction-chatmultiomics-stage123",
        "bioinstruction-chatmultiomics-stage2"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "two-number extraction and separate Pearson correlations"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "Psc requests clear, concise task answers and numeric output for regression tasks."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Section 4.2 publishes the Psc system prompt used for task performance computation.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ea-creator-systems-results"
          ],
          "metric_id": "pcc-housekeeping",
          "model_id": "bioinstruction-chatmultiomics-balanced",
          "n": 41186,
          "notes": "EA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.92
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ea-creator-systems-results"
          ],
          "metric_id": "pcc-developmental",
          "model_id": "bioinstruction-chatmultiomics-balanced",
          "n": 41186,
          "notes": "EA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.06
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ea-creator-systems-results"
          ],
          "metric_id": "pcc-housekeeping",
          "model_id": "bioinstruction-chatmultiomics-stage2",
          "n": 41186,
          "notes": "EA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.16
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ea-creator-systems-results"
          ],
          "metric_id": "pcc-developmental",
          "model_id": "bioinstruction-chatmultiomics-stage2",
          "n": 41186,
          "notes": "EA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.08
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ea-creator-systems-results"
          ],
          "metric_id": "pcc-housekeeping",
          "model_id": "bioinstruction-chatmultiomics-stage12",
          "n": 41186,
          "notes": "EA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 59.74
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ea-creator-systems-results"
          ],
          "metric_id": "pcc-developmental",
          "model_id": "bioinstruction-chatmultiomics-stage12",
          "n": 41186,
          "notes": "EA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 46.82
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ea-creator-systems-results"
          ],
          "metric_id": "pcc-housekeeping",
          "model_id": "bioinstruction-chatmultiomics-stage123",
          "n": 41186,
          "notes": "EA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 57.24
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ea-creator-systems-results"
          ],
          "metric_id": "pcc-developmental",
          "model_id": "bioinstruction-chatmultiomics-stage123",
          "n": 41186,
          "notes": "EA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 45.92
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 41186,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-ea-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-ea",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-ea-open-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-ea-open-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (EA test split); Appendix A.3; Table 8; Table 9 open-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-ea-open-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 4 (EA column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-ea-open-baselines",
      "metrics": [
        {
          "aggregation": "Pearson correlation across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "pcc-housekeeping",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "PCC (housekeeping enhancer activity)",
          "tolerance": null,
          "unit": "percent"
        },
        {
          "aggregation": "Pearson correlation across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "pcc-developmental",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "PCC (developmental enhancer activity)",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-alpaca-7b",
        "bioinstruction-galactica-13b",
        "bioinstruction-glm4-9b-chat",
        "bioinstruction-instructprotein-13b",
        "bioinstruction-llama-molinst-protein-7b",
        "bioinstruction-llama2-7b-chat",
        "bioinstruction-llama31-8b-instruct",
        "bioinstruction-qwen2-7b",
        "bioinstruction-vicuna15-7b"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "two-number extraction and separate Pearson correlations"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests the task-formatted answer and says not to explain or repeat."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes a user-prompt template and shows no system prompt for open-source models.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ea-open-baselines-results"
          ],
          "metric_id": "pcc-housekeeping",
          "model_id": "bioinstruction-llama31-8b-instruct",
          "n": 41186,
          "notes": "EA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.61
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ea-open-baselines-results"
          ],
          "metric_id": "pcc-developmental",
          "model_id": "bioinstruction-llama31-8b-instruct",
          "n": 41186,
          "notes": "EA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.27
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ea-open-baselines-results"
          ],
          "metric_id": "pcc-housekeeping",
          "model_id": "bioinstruction-qwen2-7b",
          "n": 41186,
          "notes": "EA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.4
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ea-open-baselines-results"
          ],
          "metric_id": "pcc-developmental",
          "model_id": "bioinstruction-qwen2-7b",
          "n": 41186,
          "notes": "EA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.35
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ea-open-baselines-results"
          ],
          "metric_id": "pcc-housekeeping",
          "model_id": "bioinstruction-llama2-7b-chat",
          "n": 41186,
          "notes": "EA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.55
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ea-open-baselines-results"
          ],
          "metric_id": "pcc-developmental",
          "model_id": "bioinstruction-llama2-7b-chat",
          "n": 41186,
          "notes": "EA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.13
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ea-open-baselines-results"
          ],
          "metric_id": "pcc-housekeeping",
          "model_id": "bioinstruction-alpaca-7b",
          "n": 41186,
          "notes": "EA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.11
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ea-open-baselines-results"
          ],
          "metric_id": "pcc-developmental",
          "model_id": "bioinstruction-alpaca-7b",
          "n": 41186,
          "notes": "EA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.31
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ea-open-baselines-results"
          ],
          "metric_id": "pcc-housekeeping",
          "model_id": "bioinstruction-glm4-9b-chat",
          "n": 41186,
          "notes": "EA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.87
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ea-open-baselines-results"
          ],
          "metric_id": "pcc-developmental",
          "model_id": "bioinstruction-glm4-9b-chat",
          "n": 41186,
          "notes": "EA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.17
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ea-open-baselines-results"
          ],
          "metric_id": "pcc-housekeeping",
          "model_id": "bioinstruction-vicuna15-7b",
          "n": 41186,
          "notes": "EA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.18
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ea-open-baselines-results"
          ],
          "metric_id": "pcc-developmental",
          "model_id": "bioinstruction-vicuna15-7b",
          "n": 41186,
          "notes": "EA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.69
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ea-open-baselines-results"
          ],
          "metric_id": "pcc-housekeeping",
          "model_id": "bioinstruction-galactica-13b",
          "n": 41186,
          "notes": "EA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.13
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ea-open-baselines-results"
          ],
          "metric_id": "pcc-developmental",
          "model_id": "bioinstruction-galactica-13b",
          "n": 41186,
          "notes": "EA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.09
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ea-open-baselines-results"
          ],
          "metric_id": "pcc-housekeeping",
          "model_id": "bioinstruction-instructprotein-13b",
          "n": 41186,
          "notes": "EA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ea-open-baselines-results"
          ],
          "metric_id": "pcc-developmental",
          "model_id": "bioinstruction-instructprotein-13b",
          "n": 41186,
          "notes": "EA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.39
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ea-open-baselines-results"
          ],
          "metric_id": "pcc-housekeeping",
          "model_id": "bioinstruction-llama-molinst-protein-7b",
          "n": 41186,
          "notes": "EA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.02
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ea-open-baselines-results"
          ],
          "metric_id": "pcc-developmental",
          "model_id": "bioinstruction-llama-molinst-protein-7b",
          "n": 41186,
          "notes": "EA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.1
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 41186,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-ea-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-ec",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-ec-closed-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-ec-closed-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (EC test split); Appendix A.3; Table 8; Table 9 closed-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-ec-closed-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 6 (EC column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-ec-closed-baselines",
      "metrics": [
        {
          "aggregation": "Creator Fmax implementation across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "fmax",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Fmax",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-gpt4o",
        "bioinstruction-gpt4o-mini"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "EC-number regex and multi-hot conversion followed by the creator Fmax implementation"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests a direct JSON answer and no chain-of-thought."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes the complete biology-assistant system and user prompt for closed-source models.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ec-closed-baselines-results"
          ],
          "metric_id": "fmax",
          "model_id": "bioinstruction-gpt4o-mini",
          "n": 1919,
          "notes": "EC creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 1.73
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ec-closed-baselines-results"
          ],
          "metric_id": "fmax",
          "model_id": "bioinstruction-gpt4o",
          "n": 1919,
          "notes": "EC creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 5.89
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 1919,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-ec-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-ec",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-ec-creator-systems-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-ec-creator-systems-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (EC test split); Appendix A.3; Table 8; Section 4.2 Psc prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-ec-creator-systems-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 6 (EC column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-ec-creator-systems",
      "metrics": [
        {
          "aggregation": "Creator Fmax implementation across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "fmax",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Fmax",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-chatmultiomics-balanced",
        "bioinstruction-chatmultiomics-stage12",
        "bioinstruction-chatmultiomics-stage123",
        "bioinstruction-chatmultiomics-stage2"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "EC-number regex and multi-hot conversion followed by the creator Fmax implementation"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "Psc requests clear, concise task answers and numeric output for regression tasks."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Section 4.2 publishes the Psc system prompt used for task performance computation.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ec-creator-systems-results"
          ],
          "metric_id": "fmax",
          "model_id": "bioinstruction-chatmultiomics-balanced",
          "n": 1919,
          "notes": "EC creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 10.76
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ec-creator-systems-results"
          ],
          "metric_id": "fmax",
          "model_id": "bioinstruction-chatmultiomics-stage2",
          "n": 1919,
          "notes": "EC creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 1.85
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ec-creator-systems-results"
          ],
          "metric_id": "fmax",
          "model_id": "bioinstruction-chatmultiomics-stage12",
          "n": 1919,
          "notes": "EC creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 19.35
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ec-creator-systems-results"
          ],
          "metric_id": "fmax",
          "model_id": "bioinstruction-chatmultiomics-stage123",
          "n": 1919,
          "notes": "EC creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 19.79
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 1919,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-ec-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-ec",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-ec-open-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-ec-open-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (EC test split); Appendix A.3; Table 8; Table 9 open-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-ec-open-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 6 (EC column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-ec-open-baselines",
      "metrics": [
        {
          "aggregation": "Creator Fmax implementation across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "fmax",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Fmax",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-alpaca-7b",
        "bioinstruction-biomedgpt-lm-7b",
        "bioinstruction-galactica-13b",
        "bioinstruction-glm4-9b-chat",
        "bioinstruction-instructprotein-13b",
        "bioinstruction-llama-molinst-protein-7b",
        "bioinstruction-llama2-7b-chat",
        "bioinstruction-llama31-8b-instruct",
        "bioinstruction-qwen2-7b",
        "bioinstruction-vicuna15-7b"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "EC-number regex and multi-hot conversion followed by the creator Fmax implementation"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests the task-formatted answer and says not to explain or repeat."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes a user-prompt template and shows no system prompt for open-source models.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ec-open-baselines-results"
          ],
          "metric_id": "fmax",
          "model_id": "bioinstruction-llama31-8b-instruct",
          "n": 1919,
          "notes": "EC creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 1.42
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ec-open-baselines-results"
          ],
          "metric_id": "fmax",
          "model_id": "bioinstruction-qwen2-7b",
          "n": 1919,
          "notes": "EC creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.9
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ec-open-baselines-results"
          ],
          "metric_id": "fmax",
          "model_id": "bioinstruction-llama2-7b-chat",
          "n": 1919,
          "notes": "EC creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.97
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ec-open-baselines-results"
          ],
          "metric_id": "fmax",
          "model_id": "bioinstruction-alpaca-7b",
          "n": 1919,
          "notes": "EC creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.88
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ec-open-baselines-results"
          ],
          "metric_id": "fmax",
          "model_id": "bioinstruction-glm4-9b-chat",
          "n": 1919,
          "notes": "EC creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.91
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ec-open-baselines-results"
          ],
          "metric_id": "fmax",
          "model_id": "bioinstruction-vicuna15-7b",
          "n": 1919,
          "notes": "EC creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.88
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ec-open-baselines-results"
          ],
          "metric_id": "fmax",
          "model_id": "bioinstruction-galactica-13b",
          "n": 1919,
          "notes": "EC creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.91
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ec-open-baselines-results"
          ],
          "metric_id": "fmax",
          "model_id": "bioinstruction-instructprotein-13b",
          "n": 1919,
          "notes": "EC creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 1.85
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ec-open-baselines-results"
          ],
          "metric_id": "fmax",
          "model_id": "bioinstruction-llama-molinst-protein-7b",
          "n": 1919,
          "notes": "EC creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 1.85
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ec-open-baselines-results"
          ],
          "metric_id": "fmax",
          "model_id": "bioinstruction-biomedgpt-lm-7b",
          "n": 1919,
          "notes": "EC creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 1.07
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 1919,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-ec-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-emp",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-emp-closed-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-emp-closed-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (EMP test split); Appendix A.3; Table 8; Table 9 closed-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-emp-closed-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 4 (EMP column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-emp-closed-baselines",
      "metrics": [
        {
          "aggregation": "Matthews correlation coefficient across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "mcc",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "MCC",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-gpt4o",
        "bioinstruction-gpt4o-mini"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": "cardiffnlp/twitter-roberta-base-sentiment-latest fallback",
          "reporting_status": "reported",
          "type": "keyword-first binary extraction with sentiment-model fallback, followed by MCC"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests a direct JSON answer and no chain-of-thought."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes the complete biology-assistant system and user prompt for closed-source models.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-emp-closed-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-gpt4o-mini",
          "n": 28741,
          "notes": "EMP creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.91
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-emp-closed-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-gpt4o",
          "n": 28741,
          "notes": "EMP creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.49
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 28741,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-emp-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-emp",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-emp-creator-systems-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-emp-creator-systems-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (EMP test split); Appendix A.3; Table 8; Section 4.2 Psc prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-emp-creator-systems-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 4 (EMP column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-emp-creator-systems",
      "metrics": [
        {
          "aggregation": "Matthews correlation coefficient across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "mcc",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "MCC",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-chatmultiomics-balanced",
        "bioinstruction-chatmultiomics-stage12",
        "bioinstruction-chatmultiomics-stage123",
        "bioinstruction-chatmultiomics-stage2"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": "cardiffnlp/twitter-roberta-base-sentiment-latest fallback",
          "reporting_status": "reported",
          "type": "keyword-first binary extraction with sentiment-model fallback, followed by MCC"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "Psc requests clear, concise task answers and numeric output for regression tasks."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Section 4.2 publishes the Psc system prompt used for task performance computation.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-emp-creator-systems-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-chatmultiomics-balanced",
          "n": 28741,
          "notes": "EMP creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 1.4
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-emp-creator-systems-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-chatmultiomics-stage2",
          "n": 28741,
          "notes": "EMP creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.31
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-emp-creator-systems-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-chatmultiomics-stage12",
          "n": 28741,
          "notes": "EMP creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 8.1
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-emp-creator-systems-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-chatmultiomics-stage123",
          "n": 28741,
          "notes": "EMP creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 3.64
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 28741,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-emp-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-emp",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-emp-open-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-emp-open-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (EMP test split); Appendix A.3; Table 8; Table 9 open-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-emp-open-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 4 (EMP column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-emp-open-baselines",
      "metrics": [
        {
          "aggregation": "Matthews correlation coefficient across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "mcc",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "MCC",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-alpaca-7b",
        "bioinstruction-galactica-13b",
        "bioinstruction-glm4-9b-chat",
        "bioinstruction-instructprotein-13b",
        "bioinstruction-llama-molinst-protein-7b",
        "bioinstruction-llama2-7b-chat",
        "bioinstruction-llama31-8b-instruct",
        "bioinstruction-qwen2-7b",
        "bioinstruction-vicuna15-7b"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": "cardiffnlp/twitter-roberta-base-sentiment-latest fallback",
          "reporting_status": "reported",
          "type": "keyword-first binary extraction with sentiment-model fallback, followed by MCC"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests the task-formatted answer and says not to explain or repeat."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes a user-prompt template and shows no system prompt for open-source models.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-emp-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-llama31-8b-instruct",
          "n": 28741,
          "notes": "EMP creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.37
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-emp-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-qwen2-7b",
          "n": 28741,
          "notes": "EMP creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.66
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-emp-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-llama2-7b-chat",
          "n": 28741,
          "notes": "EMP creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.94
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-emp-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-alpaca-7b",
          "n": 28741,
          "notes": "EMP creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.36
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-emp-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-glm4-9b-chat",
          "n": 28741,
          "notes": "EMP creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.22
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-emp-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-vicuna15-7b",
          "n": 28741,
          "notes": "EMP creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-emp-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-galactica-13b",
          "n": 28741,
          "notes": "EMP creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.07
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-emp-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-instructprotein-13b",
          "n": 28741,
          "notes": "EMP creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.22
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-emp-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-llama-molinst-protein-7b",
          "n": 28741,
          "notes": "EMP creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.29
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 28741,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-emp-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-epi",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-epi-closed-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-epi-closed-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (EPI test split); Appendix A.3; Table 8; Table 9 closed-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-epi-closed-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 7 (EPI column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-epi-closed-baselines",
      "metrics": [
        {
          "aggregation": "Matthews correlation coefficient across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "mcc",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "MCC",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-gpt4o",
        "bioinstruction-gpt4o-mini"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": "cardiffnlp/twitter-roberta-base-sentiment-latest fallback",
          "reporting_status": "reported",
          "type": "keyword-first binary extraction with sentiment-model fallback, followed by MCC"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests a direct JSON answer and no chain-of-thought."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes the complete biology-assistant system and user prompt for closed-source models.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-epi-closed-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-gpt4o-mini",
          "n": 308,
          "notes": "EPI creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.39
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-epi-closed-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-gpt4o",
          "n": 308,
          "notes": "EPI creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 308,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-epi-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-epi",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-epi-creator-systems-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-epi-creator-systems-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (EPI test split); Appendix A.3; Table 8; Section 4.2 Psc prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-epi-creator-systems-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 7 (EPI column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-epi-creator-systems",
      "metrics": [
        {
          "aggregation": "Matthews correlation coefficient across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "mcc",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "MCC",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-chatmultiomics-balanced",
        "bioinstruction-chatmultiomics-stage12",
        "bioinstruction-chatmultiomics-stage123",
        "bioinstruction-chatmultiomics-stage2"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": "cardiffnlp/twitter-roberta-base-sentiment-latest fallback",
          "reporting_status": "reported",
          "type": "keyword-first binary extraction with sentiment-model fallback, followed by MCC"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "Psc requests clear, concise task answers and numeric output for regression tasks."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Section 4.2 publishes the Psc system prompt used for task performance computation.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-epi-creator-systems-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-chatmultiomics-balanced",
          "n": 308,
          "notes": "EPI creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 4.13
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-epi-creator-systems-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-chatmultiomics-stage2",
          "n": 308,
          "notes": "EPI creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 4.77
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-epi-creator-systems-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-chatmultiomics-stage12",
          "n": 308,
          "notes": "EPI creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 1.68
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-epi-creator-systems-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-chatmultiomics-stage123",
          "n": 308,
          "notes": "EPI creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 3.37
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 308,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-epi-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-epi",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-epi-open-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-epi-open-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (EPI test split); Appendix A.3; Table 8; Table 9 open-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-epi-open-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 7 (EPI column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-epi-open-baselines",
      "metrics": [
        {
          "aggregation": "Matthews correlation coefficient across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "mcc",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "MCC",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-alpaca-7b",
        "bioinstruction-biomedgpt-lm-7b",
        "bioinstruction-galactica-13b",
        "bioinstruction-glm4-9b-chat",
        "bioinstruction-instructprotein-13b",
        "bioinstruction-llama-molinst-protein-7b",
        "bioinstruction-llama2-7b-chat",
        "bioinstruction-llama31-8b-instruct",
        "bioinstruction-qwen2-7b",
        "bioinstruction-vicuna15-7b"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": "cardiffnlp/twitter-roberta-base-sentiment-latest fallback",
          "reporting_status": "reported",
          "type": "keyword-first binary extraction with sentiment-model fallback, followed by MCC"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests the task-formatted answer and says not to explain or repeat."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes a user-prompt template and shows no system prompt for open-source models.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-epi-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-llama31-8b-instruct",
          "n": 308,
          "notes": "EPI creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-epi-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-qwen2-7b",
          "n": 308,
          "notes": "EPI creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-epi-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-llama2-7b-chat",
          "n": 308,
          "notes": "EPI creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-epi-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-alpaca-7b",
          "n": 308,
          "notes": "EPI creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-epi-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-glm4-9b-chat",
          "n": 308,
          "notes": "EPI creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-epi-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-vicuna15-7b",
          "n": 308,
          "notes": "EPI creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-epi-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-galactica-13b",
          "n": 308,
          "notes": "EPI creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-epi-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-instructprotein-13b",
          "n": 308,
          "notes": "EPI creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-epi-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-llama-molinst-protein-7b",
          "n": 308,
          "notes": "EPI creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-epi-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-biomedgpt-lm-7b",
          "n": 308,
          "notes": "EPI creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 308,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-epi-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-fluorescence",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-fluorescence-closed-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-fluorescence-closed-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (Flu test split); Appendix A.3; Table 8; Table 9 closed-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-fluorescence-closed-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 6 (Flu column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-fluorescence-closed-baselines",
      "metrics": [
        {
          "aggregation": "Spearman rank correlation across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "spearman-rho",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "Spearman's ρ",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-gpt4o",
        "bioinstruction-gpt4o-mini"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "first-number extraction followed by Spearman rank correlation"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests a direct JSON answer and no chain-of-thought."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes the complete biology-assistant system and user prompt for closed-source models.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-fluorescence-closed-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-gpt4o-mini",
          "n": 27217,
          "notes": "Flu creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.47
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-fluorescence-closed-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-gpt4o",
          "n": 27217,
          "notes": "Flu creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.69
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 27217,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-fluorescence-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-fluorescence",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-fluorescence-creator-systems-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-fluorescence-creator-systems-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (Flu test split); Appendix A.3; Table 8; Section 4.2 Psc prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-fluorescence-creator-systems-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 6 (Flu column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-fluorescence-creator-systems",
      "metrics": [
        {
          "aggregation": "Spearman rank correlation across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "spearman-rho",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "Spearman's ρ",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-chatmultiomics-balanced",
        "bioinstruction-chatmultiomics-stage12",
        "bioinstruction-chatmultiomics-stage123",
        "bioinstruction-chatmultiomics-stage2"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "first-number extraction followed by Spearman rank correlation"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "Psc requests clear, concise task answers and numeric output for regression tasks."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Section 4.2 publishes the Psc system prompt used for task performance computation.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-fluorescence-creator-systems-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-chatmultiomics-balanced",
          "n": 27217,
          "notes": "Flu creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.55
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-fluorescence-creator-systems-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-chatmultiomics-stage2",
          "n": 27217,
          "notes": "Flu creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.37
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-fluorescence-creator-systems-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-chatmultiomics-stage12",
          "n": 27217,
          "notes": "Flu creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 1.49
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-fluorescence-creator-systems-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-chatmultiomics-stage123",
          "n": 27217,
          "notes": "Flu creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 2.57
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 27217,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-fluorescence-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-fluorescence",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-fluorescence-open-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-fluorescence-open-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (Flu test split); Appendix A.3; Table 8; Table 9 open-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-fluorescence-open-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 6 (Flu column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-fluorescence-open-baselines",
      "metrics": [
        {
          "aggregation": "Spearman rank correlation across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "spearman-rho",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "Spearman's ρ",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-alpaca-7b",
        "bioinstruction-biomedgpt-lm-7b",
        "bioinstruction-galactica-13b",
        "bioinstruction-glm4-9b-chat",
        "bioinstruction-instructprotein-13b",
        "bioinstruction-llama-molinst-protein-7b",
        "bioinstruction-llama2-7b-chat",
        "bioinstruction-llama31-8b-instruct",
        "bioinstruction-qwen2-7b",
        "bioinstruction-vicuna15-7b"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "first-number extraction followed by Spearman rank correlation"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests the task-formatted answer and says not to explain or repeat."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes a user-prompt template and shows no system prompt for open-source models.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-fluorescence-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-llama31-8b-instruct",
          "n": 27217,
          "notes": "Flu creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.91
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-fluorescence-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-qwen2-7b",
          "n": 27217,
          "notes": "Flu creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.81
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-fluorescence-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-llama2-7b-chat",
          "n": 27217,
          "notes": "Flu creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.28
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-fluorescence-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-alpaca-7b",
          "n": 27217,
          "notes": "Flu creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.2
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-fluorescence-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-glm4-9b-chat",
          "n": 27217,
          "notes": "Flu creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.63
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-fluorescence-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-vicuna15-7b",
          "n": 27217,
          "notes": "Flu creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.51
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-fluorescence-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-galactica-13b",
          "n": 27217,
          "notes": "Flu creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.73
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-fluorescence-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-instructprotein-13b",
          "n": 27217,
          "notes": "Flu creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.03
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-fluorescence-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-llama-molinst-protein-7b",
          "n": 27217,
          "notes": "Flu creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.27
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-fluorescence-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-biomedgpt-lm-7b",
          "n": 27217,
          "notes": "Flu creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.43
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 27217,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-fluorescence-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-modification",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-modification-closed-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-modification-closed-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (Modif test split); Appendix A.3; Table 8; Table 9 closed-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-modification-closed-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 5 (Modif column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-modification-closed-baselines",
      "metrics": [
        {
          "aggregation": "Macro ROC AUC across modification labels on held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "auc",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "AUC",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-gpt4o",
        "bioinstruction-gpt4o-mini"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": "cardiffnlp/twitter-roberta-base-sentiment-latest fallback",
          "reporting_status": "reported",
          "type": "modification-label extraction with sentiment fallback for none, followed by macro ROC AUC"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests a direct JSON answer and no chain-of-thought."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes the complete biology-assistant system and user prompt for closed-source models.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-modification-closed-baselines-results"
          ],
          "metric_id": "auc",
          "model_id": "bioinstruction-gpt4o-mini",
          "n": 1200,
          "notes": "Modif creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 50.49
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-modification-closed-baselines-results"
          ],
          "metric_id": "auc",
          "model_id": "bioinstruction-gpt4o",
          "n": 1200,
          "notes": "Modif creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 50.47
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 1200,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-modification-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-modification",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-modification-creator-systems-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-modification-creator-systems-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (Modif test split); Appendix A.3; Table 8; Section 4.2 Psc prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-modification-creator-systems-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 5 (Modif column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-modification-creator-systems",
      "metrics": [
        {
          "aggregation": "Macro ROC AUC across modification labels on held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "auc",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "AUC",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-chatmultiomics-balanced",
        "bioinstruction-chatmultiomics-stage12",
        "bioinstruction-chatmultiomics-stage123",
        "bioinstruction-chatmultiomics-stage2"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": "cardiffnlp/twitter-roberta-base-sentiment-latest fallback",
          "reporting_status": "reported",
          "type": "modification-label extraction with sentiment fallback for none, followed by macro ROC AUC"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "Psc requests clear, concise task answers and numeric output for regression tasks."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Section 4.2 publishes the Psc system prompt used for task performance computation.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-modification-creator-systems-results"
          ],
          "metric_id": "auc",
          "model_id": "bioinstruction-chatmultiomics-balanced",
          "n": 1200,
          "notes": "Modif creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 53.76
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-modification-creator-systems-results"
          ],
          "metric_id": "auc",
          "model_id": "bioinstruction-chatmultiomics-stage2",
          "n": 1200,
          "notes": "Modif creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 51.21
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-modification-creator-systems-results"
          ],
          "metric_id": "auc",
          "model_id": "bioinstruction-chatmultiomics-stage12",
          "n": 1200,
          "notes": "Modif creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 57.45
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-modification-creator-systems-results"
          ],
          "metric_id": "auc",
          "model_id": "bioinstruction-chatmultiomics-stage123",
          "n": 1200,
          "notes": "Modif creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 59.06
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 1200,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-modification-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-modification",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-modification-open-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-modification-open-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (Modif test split); Appendix A.3; Table 8; Table 9 open-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-modification-open-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 5 (Modif column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-modification-open-baselines",
      "metrics": [
        {
          "aggregation": "Macro ROC AUC across modification labels on held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "auc",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "AUC",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-alpaca-7b",
        "bioinstruction-biomedgpt-lm-7b",
        "bioinstruction-galactica-13b",
        "bioinstruction-glm4-9b-chat",
        "bioinstruction-instructprotein-13b",
        "bioinstruction-llama-molinst-protein-7b",
        "bioinstruction-llama2-7b-chat",
        "bioinstruction-llama31-8b-instruct",
        "bioinstruction-qwen2-7b",
        "bioinstruction-vicuna15-7b"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": "cardiffnlp/twitter-roberta-base-sentiment-latest fallback",
          "reporting_status": "reported",
          "type": "modification-label extraction with sentiment fallback for none, followed by macro ROC AUC"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests the task-formatted answer and says not to explain or repeat."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes a user-prompt template and shows no system prompt for open-source models.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-modification-open-baselines-results"
          ],
          "metric_id": "auc",
          "model_id": "bioinstruction-llama31-8b-instruct",
          "n": 1200,
          "notes": "Modif creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 50.52
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-modification-open-baselines-results"
          ],
          "metric_id": "auc",
          "model_id": "bioinstruction-qwen2-7b",
          "n": 1200,
          "notes": "Modif creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 50.34
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-modification-open-baselines-results"
          ],
          "metric_id": "auc",
          "model_id": "bioinstruction-llama2-7b-chat",
          "n": 1200,
          "notes": "Modif creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 50.4
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-modification-open-baselines-results"
          ],
          "metric_id": "auc",
          "model_id": "bioinstruction-alpaca-7b",
          "n": 1200,
          "notes": "Modif creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 50
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-modification-open-baselines-results"
          ],
          "metric_id": "auc",
          "model_id": "bioinstruction-glm4-9b-chat",
          "n": 1200,
          "notes": "Modif creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 50.05
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-modification-open-baselines-results"
          ],
          "metric_id": "auc",
          "model_id": "bioinstruction-vicuna15-7b",
          "n": 1200,
          "notes": "Modif creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 50.27
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-modification-open-baselines-results"
          ],
          "metric_id": "auc",
          "model_id": "bioinstruction-galactica-13b",
          "n": 1200,
          "notes": "Modif creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 53.78
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-modification-open-baselines-results"
          ],
          "metric_id": "auc",
          "model_id": "bioinstruction-instructprotein-13b",
          "n": 1200,
          "notes": "Modif creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 51.08
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-modification-open-baselines-results"
          ],
          "metric_id": "auc",
          "model_id": "bioinstruction-llama-molinst-protein-7b",
          "n": 1200,
          "notes": "Modif creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 52.51
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-modification-open-baselines-results"
          ],
          "metric_id": "auc",
          "model_id": "bioinstruction-biomedgpt-lm-7b",
          "n": 1200,
          "notes": "Modif creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 51.65
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 1200,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-modification-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-mrl",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-mrl-closed-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-mrl-closed-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (MRL test split); Appendix A.3; Table 8; Table 9 closed-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-mrl-closed-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 5 (MRL column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-mrl-closed-baselines",
      "metrics": [
        {
          "aggregation": "Squared Pearson correlation across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "r2",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "R2",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-gpt4o",
        "bioinstruction-gpt4o-mini"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "first-number extraction followed by squared Pearson correlation labeled R2"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests a direct JSON answer and no chain-of-thought."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes the complete biology-assistant system and user prompt for closed-source models.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-mrl-closed-baselines-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-gpt4o-mini",
          "n": 7600,
          "notes": "MRL creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.01
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-mrl-closed-baselines-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-gpt4o",
          "n": 7600,
          "notes": "MRL creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.01
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 7600,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-mrl-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-mrl",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-mrl-creator-systems-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-mrl-creator-systems-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (MRL test split); Appendix A.3; Table 8; Section 4.2 Psc prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-mrl-creator-systems-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 5 (MRL column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-mrl-creator-systems",
      "metrics": [
        {
          "aggregation": "Squared Pearson correlation across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "r2",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "R2",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-chatmultiomics-balanced",
        "bioinstruction-chatmultiomics-stage12",
        "bioinstruction-chatmultiomics-stage123",
        "bioinstruction-chatmultiomics-stage2"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "first-number extraction followed by squared Pearson correlation labeled R2"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "Psc requests clear, concise task answers and numeric output for regression tasks."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Section 4.2 publishes the Psc system prompt used for task performance computation.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-mrl-creator-systems-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-chatmultiomics-balanced",
          "n": 7600,
          "notes": "MRL creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-mrl-creator-systems-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-chatmultiomics-stage2",
          "n": 7600,
          "notes": "MRL creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-mrl-creator-systems-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-chatmultiomics-stage12",
          "n": 7600,
          "notes": "MRL creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 29.12
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-mrl-creator-systems-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-chatmultiomics-stage123",
          "n": 7600,
          "notes": "MRL creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 47.64
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 7600,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-mrl-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-mrl",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-mrl-open-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-mrl-open-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (MRL test split); Appendix A.3; Table 8; Table 9 open-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-mrl-open-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 5 (MRL column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-mrl-open-baselines",
      "metrics": [
        {
          "aggregation": "Squared Pearson correlation across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "r2",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "R2",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-alpaca-7b",
        "bioinstruction-biomedgpt-lm-7b",
        "bioinstruction-galactica-13b",
        "bioinstruction-glm4-9b-chat",
        "bioinstruction-instructprotein-13b",
        "bioinstruction-llama-molinst-protein-7b",
        "bioinstruction-llama2-7b-chat",
        "bioinstruction-llama31-8b-instruct",
        "bioinstruction-qwen2-7b",
        "bioinstruction-vicuna15-7b"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "first-number extraction followed by squared Pearson correlation labeled R2"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests the task-formatted answer and says not to explain or repeat."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes a user-prompt template and shows no system prompt for open-source models.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-mrl-open-baselines-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-llama31-8b-instruct",
          "n": 7600,
          "notes": "MRL creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.01
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-mrl-open-baselines-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-qwen2-7b",
          "n": 7600,
          "notes": "MRL creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-mrl-open-baselines-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-llama2-7b-chat",
          "n": 7600,
          "notes": "MRL creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-mrl-open-baselines-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-alpaca-7b",
          "n": 7600,
          "notes": "MRL creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.03
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-mrl-open-baselines-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-glm4-9b-chat",
          "n": 7600,
          "notes": "MRL creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-mrl-open-baselines-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-vicuna15-7b",
          "n": 7600,
          "notes": "MRL creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.01
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-mrl-open-baselines-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-galactica-13b",
          "n": 7600,
          "notes": "MRL creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-mrl-open-baselines-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-instructprotein-13b",
          "n": 7600,
          "notes": "MRL creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.02
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-mrl-open-baselines-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-llama-molinst-protein-7b",
          "n": 7600,
          "notes": "MRL creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-mrl-open-baselines-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-biomedgpt-lm-7b",
          "n": 7600,
          "notes": "MRL creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.01
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 7600,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-mrl-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-ncrna",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-ncrna-closed-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-ncrna-closed-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (ncRNA test split); Appendix A.3; Table 8; Table 9 closed-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-ncrna-closed-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 5 (ncRNA column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-ncrna-closed-baselines",
      "metrics": [
        {
          "aggregation": "Exact extracted-class accuracy across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Accuracy",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-gpt4o",
        "bioinstruction-gpt4o-mini"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "ordered RNA-family name extraction followed by exact accuracy"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests a direct JSON answer and no chain-of-thought."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes the complete biology-assistant system and user prompt for closed-source models.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ncrna-closed-baselines-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bioinstruction-gpt4o-mini",
          "n": 4840,
          "notes": "ncRNA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 3
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ncrna-closed-baselines-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bioinstruction-gpt4o",
          "n": 4840,
          "notes": "ncRNA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 5.6
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 4840,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-ncrna-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-ncrna",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-ncrna-creator-systems-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-ncrna-creator-systems-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (ncRNA test split); Appendix A.3; Table 8; Section 4.2 Psc prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-ncrna-creator-systems-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 5 (ncRNA column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-ncrna-creator-systems",
      "metrics": [
        {
          "aggregation": "Exact extracted-class accuracy across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Accuracy",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-chatmultiomics-balanced",
        "bioinstruction-chatmultiomics-stage12",
        "bioinstruction-chatmultiomics-stage123",
        "bioinstruction-chatmultiomics-stage2"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "ordered RNA-family name extraction followed by exact accuracy"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "Psc requests clear, concise task answers and numeric output for regression tasks."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Section 4.2 publishes the Psc system prompt used for task performance computation.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ncrna-creator-systems-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bioinstruction-chatmultiomics-balanced",
          "n": 4840,
          "notes": "ncRNA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 35.68
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ncrna-creator-systems-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bioinstruction-chatmultiomics-stage2",
          "n": 4840,
          "notes": "ncRNA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ncrna-creator-systems-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bioinstruction-chatmultiomics-stage12",
          "n": 4840,
          "notes": "ncRNA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 62.77
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ncrna-creator-systems-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bioinstruction-chatmultiomics-stage123",
          "n": 4840,
          "notes": "ncRNA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 63.09
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 4840,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-ncrna-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-ncrna",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-ncrna-open-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-ncrna-open-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (ncRNA test split); Appendix A.3; Table 8; Table 9 open-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-ncrna-open-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 5 (ncRNA column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-ncrna-open-baselines",
      "metrics": [
        {
          "aggregation": "Exact extracted-class accuracy across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Accuracy",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-alpaca-7b",
        "bioinstruction-biomedgpt-lm-7b",
        "bioinstruction-galactica-13b",
        "bioinstruction-glm4-9b-chat",
        "bioinstruction-instructprotein-13b",
        "bioinstruction-llama-molinst-protein-7b",
        "bioinstruction-llama2-7b-chat",
        "bioinstruction-llama31-8b-instruct",
        "bioinstruction-qwen2-7b",
        "bioinstruction-vicuna15-7b"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "ordered RNA-family name extraction followed by exact accuracy"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests the task-formatted answer and says not to explain or repeat."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes a user-prompt template and shows no system prompt for open-source models.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ncrna-open-baselines-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bioinstruction-llama31-8b-instruct",
          "n": 4840,
          "notes": "ncRNA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 6.32
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ncrna-open-baselines-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bioinstruction-qwen2-7b",
          "n": 4840,
          "notes": "ncRNA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 7.08
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ncrna-open-baselines-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bioinstruction-llama2-7b-chat",
          "n": 4840,
          "notes": "ncRNA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 4.88
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ncrna-open-baselines-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bioinstruction-alpaca-7b",
          "n": 4840,
          "notes": "ncRNA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 7.42
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ncrna-open-baselines-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bioinstruction-glm4-9b-chat",
          "n": 4840,
          "notes": "ncRNA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 8.23
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ncrna-open-baselines-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bioinstruction-vicuna15-7b",
          "n": 4840,
          "notes": "ncRNA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 3.81
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ncrna-open-baselines-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bioinstruction-galactica-13b",
          "n": 4840,
          "notes": "ncRNA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 6.73
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ncrna-open-baselines-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bioinstruction-instructprotein-13b",
          "n": 4840,
          "notes": "ncRNA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ncrna-open-baselines-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bioinstruction-llama-molinst-protein-7b",
          "n": 4840,
          "notes": "ncRNA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-ncrna-open-baselines-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bioinstruction-biomedgpt-lm-7b",
          "n": 4840,
          "notes": "ncRNA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 1.62
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 4840,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-ncrna-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-pd300",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-pd300-closed-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-pd300-closed-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (PD300 test split); Appendix A.3; Table 8; Table 9 closed-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-pd300-closed-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 4 (PD300 column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-pd300-closed-baselines",
      "metrics": [
        {
          "aggregation": "Matthews correlation coefficient across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "mcc",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "MCC",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-gpt4o",
        "bioinstruction-gpt4o-mini"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": "cardiffnlp/twitter-roberta-base-sentiment-latest fallback",
          "reporting_status": "reported",
          "type": "keyword-first binary extraction with sentiment-model fallback, followed by MCC"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests a direct JSON answer and no chain-of-thought."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes the complete biology-assistant system and user prompt for closed-source models.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-pd300-closed-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-gpt4o-mini",
          "n": 11840,
          "notes": "PD300 creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -4.44
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-pd300-closed-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-gpt4o",
          "n": 11840,
          "notes": "PD300 creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 8.67
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 11840,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-pd300-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-pd300",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-pd300-creator-systems-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-pd300-creator-systems-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (PD300 test split); Appendix A.3; Table 8; Section 4.2 Psc prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-pd300-creator-systems-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 4 (PD300 column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-pd300-creator-systems",
      "metrics": [
        {
          "aggregation": "Matthews correlation coefficient across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "mcc",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "MCC",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-chatmultiomics-balanced",
        "bioinstruction-chatmultiomics-stage12",
        "bioinstruction-chatmultiomics-stage123",
        "bioinstruction-chatmultiomics-stage2"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": "cardiffnlp/twitter-roberta-base-sentiment-latest fallback",
          "reporting_status": "reported",
          "type": "keyword-first binary extraction with sentiment-model fallback, followed by MCC"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "Psc requests clear, concise task answers and numeric output for regression tasks."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Section 4.2 publishes the Psc system prompt used for task performance computation.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-pd300-creator-systems-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-chatmultiomics-balanced",
          "n": 11840,
          "notes": "PD300 creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 5.19
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-pd300-creator-systems-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-chatmultiomics-stage2",
          "n": 11840,
          "notes": "PD300 creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.87
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-pd300-creator-systems-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-chatmultiomics-stage12",
          "n": 11840,
          "notes": "PD300 creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 49.01
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-pd300-creator-systems-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-chatmultiomics-stage123",
          "n": 11840,
          "notes": "PD300 creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 58.18
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 11840,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-pd300-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-pd300",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-pd300-open-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-pd300-open-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (PD300 test split); Appendix A.3; Table 8; Table 9 open-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-pd300-open-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 4 (PD300 column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-pd300-open-baselines",
      "metrics": [
        {
          "aggregation": "Matthews correlation coefficient across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "mcc",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "MCC",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-alpaca-7b",
        "bioinstruction-galactica-13b",
        "bioinstruction-glm4-9b-chat",
        "bioinstruction-instructprotein-13b",
        "bioinstruction-llama-molinst-protein-7b",
        "bioinstruction-llama2-7b-chat",
        "bioinstruction-llama31-8b-instruct",
        "bioinstruction-qwen2-7b",
        "bioinstruction-vicuna15-7b"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": "cardiffnlp/twitter-roberta-base-sentiment-latest fallback",
          "reporting_status": "reported",
          "type": "keyword-first binary extraction with sentiment-model fallback, followed by MCC"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests the task-formatted answer and says not to explain or repeat."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes a user-prompt template and shows no system prompt for open-source models.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-pd300-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-llama31-8b-instruct",
          "n": 11840,
          "notes": "PD300 creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.01
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-pd300-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-qwen2-7b",
          "n": 11840,
          "notes": "PD300 creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -4.83
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-pd300-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-llama2-7b-chat",
          "n": 11840,
          "notes": "PD300 creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.29
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-pd300-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-alpaca-7b",
          "n": 11840,
          "notes": "PD300 creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.15
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-pd300-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-glm4-9b-chat",
          "n": 11840,
          "notes": "PD300 creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.25
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-pd300-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-vicuna15-7b",
          "n": 11840,
          "notes": "PD300 creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-pd300-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-galactica-13b",
          "n": 11840,
          "notes": "PD300 creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.41
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-pd300-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-instructprotein-13b",
          "n": 11840,
          "notes": "PD300 creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 2.75
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-pd300-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-llama-molinst-protein-7b",
          "n": 11840,
          "notes": "PD300 creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -5.76
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 11840,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-pd300-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-prs",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-prs-closed-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-prs-closed-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (PRS test split); Appendix A.3; Table 8; Table 9 closed-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-prs-closed-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 5 (PRS column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-prs-closed-baselines",
      "metrics": [
        {
          "aggregation": "Mean of ON, OFF, and ON/OFF squared Pearson correlations, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "r2",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "R2",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-gpt4o",
        "bioinstruction-gpt4o-mini"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "three-number extraction followed by the mean of three squared Pearson correlations"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests a direct JSON answer and no chain-of-thought."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes the complete biology-assistant system and user prompt for closed-source models.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-prs-closed-baselines-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-gpt4o-mini",
          "n": 11019,
          "notes": "PRS creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.03
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-prs-closed-baselines-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-gpt4o",
          "n": 11019,
          "notes": "PRS creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 11019,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-prs-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-prs",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-prs-creator-systems-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-prs-creator-systems-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (PRS test split); Appendix A.3; Table 8; Section 4.2 Psc prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-prs-creator-systems-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 5 (PRS column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-prs-creator-systems",
      "metrics": [
        {
          "aggregation": "Mean of ON, OFF, and ON/OFF squared Pearson correlations, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "r2",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "R2",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-chatmultiomics-balanced",
        "bioinstruction-chatmultiomics-stage12",
        "bioinstruction-chatmultiomics-stage123",
        "bioinstruction-chatmultiomics-stage2"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "three-number extraction followed by the mean of three squared Pearson correlations"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "Psc requests clear, concise task answers and numeric output for regression tasks."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Section 4.2 publishes the Psc system prompt used for task performance computation.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-prs-creator-systems-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-chatmultiomics-balanced",
          "n": 11019,
          "notes": "PRS creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.01
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-prs-creator-systems-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-chatmultiomics-stage2",
          "n": 11019,
          "notes": "PRS creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-prs-creator-systems-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-chatmultiomics-stage12",
          "n": 11019,
          "notes": "PRS creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 26.65
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-prs-creator-systems-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-chatmultiomics-stage123",
          "n": 11019,
          "notes": "PRS creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 26.57
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 11019,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-prs-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-prs",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-prs-open-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-prs-open-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (PRS test split); Appendix A.3; Table 8; Table 9 open-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-prs-open-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 5 (PRS column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-prs-open-baselines",
      "metrics": [
        {
          "aggregation": "Mean of ON, OFF, and ON/OFF squared Pearson correlations, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "r2",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "R2",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-alpaca-7b",
        "bioinstruction-biomedgpt-lm-7b",
        "bioinstruction-galactica-13b",
        "bioinstruction-glm4-9b-chat",
        "bioinstruction-instructprotein-13b",
        "bioinstruction-llama-molinst-protein-7b",
        "bioinstruction-llama2-7b-chat",
        "bioinstruction-llama31-8b-instruct",
        "bioinstruction-qwen2-7b",
        "bioinstruction-vicuna15-7b"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "three-number extraction followed by the mean of three squared Pearson correlations"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests the task-formatted answer and says not to explain or repeat."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes a user-prompt template and shows no system prompt for open-source models.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-prs-open-baselines-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-llama31-8b-instruct",
          "n": 11019,
          "notes": "PRS creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.02
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-prs-open-baselines-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-qwen2-7b",
          "n": 11019,
          "notes": "PRS creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.01
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-prs-open-baselines-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-llama2-7b-chat",
          "n": 11019,
          "notes": "PRS creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.01
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-prs-open-baselines-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-alpaca-7b",
          "n": 11019,
          "notes": "PRS creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.01
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-prs-open-baselines-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-glm4-9b-chat",
          "n": 11019,
          "notes": "PRS creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.01
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-prs-open-baselines-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-vicuna15-7b",
          "n": 11019,
          "notes": "PRS creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-prs-open-baselines-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-galactica-13b",
          "n": 11019,
          "notes": "PRS creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.02
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-prs-open-baselines-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-instructprotein-13b",
          "n": 11019,
          "notes": "PRS creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-prs-open-baselines-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-llama-molinst-protein-7b",
          "n": 11019,
          "notes": "PRS creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.02
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-prs-open-baselines-results"
          ],
          "metric_id": "r2",
          "model_id": "bioinstruction-biomedgpt-lm-7b",
          "n": 11019,
          "notes": "PRS creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.03
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 11019,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-prs-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-rpi",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-rpi-closed-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-rpi-closed-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (RPI test split); Appendix A.3; Table 8; Table 9 closed-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-rpi-closed-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 7 (RPI column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-rpi-closed-baselines",
      "metrics": [
        {
          "aggregation": "Matthews correlation coefficient across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "mcc",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "MCC",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-gpt4o",
        "bioinstruction-gpt4o-mini"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": "cardiffnlp/twitter-roberta-base-sentiment-latest fallback",
          "reporting_status": "reported",
          "type": "keyword-first binary extraction with sentiment-model fallback, followed by MCC"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests a direct JSON answer and no chain-of-thought."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes the complete biology-assistant system and user prompt for closed-source models.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-rpi-closed-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-gpt4o-mini",
          "n": 4164,
          "notes": "RPI creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 1.22
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-rpi-closed-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-gpt4o",
          "n": 4164,
          "notes": "RPI creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 1.17
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 4164,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-rpi-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-rpi",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-rpi-creator-systems-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-rpi-creator-systems-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (RPI test split); Appendix A.3; Table 8; Section 4.2 Psc prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-rpi-creator-systems-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 7 (RPI column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-rpi-creator-systems",
      "metrics": [
        {
          "aggregation": "Matthews correlation coefficient across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "mcc",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "MCC",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-chatmultiomics-balanced",
        "bioinstruction-chatmultiomics-stage12",
        "bioinstruction-chatmultiomics-stage123",
        "bioinstruction-chatmultiomics-stage2"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": "cardiffnlp/twitter-roberta-base-sentiment-latest fallback",
          "reporting_status": "reported",
          "type": "keyword-first binary extraction with sentiment-model fallback, followed by MCC"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "Psc requests clear, concise task answers and numeric output for regression tasks."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Section 4.2 publishes the Psc system prompt used for task performance computation.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-rpi-creator-systems-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-chatmultiomics-balanced",
          "n": 4164,
          "notes": "RPI creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 8.29
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-rpi-creator-systems-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-chatmultiomics-stage2",
          "n": 4164,
          "notes": "RPI creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 1.61
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-rpi-creator-systems-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-chatmultiomics-stage12",
          "n": 4164,
          "notes": "RPI creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 70.8
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-rpi-creator-systems-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-chatmultiomics-stage123",
          "n": 4164,
          "notes": "RPI creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 74.26
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 4164,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-rpi-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-rpi",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-rpi-open-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-rpi-open-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (RPI test split); Appendix A.3; Table 8; Table 9 open-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-rpi-open-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 7 (RPI column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-rpi-open-baselines",
      "metrics": [
        {
          "aggregation": "Matthews correlation coefficient across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "mcc",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "MCC",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-alpaca-7b",
        "bioinstruction-biomedgpt-lm-7b",
        "bioinstruction-galactica-13b",
        "bioinstruction-glm4-9b-chat",
        "bioinstruction-instructprotein-13b",
        "bioinstruction-llama-molinst-protein-7b",
        "bioinstruction-llama2-7b-chat",
        "bioinstruction-llama31-8b-instruct",
        "bioinstruction-qwen2-7b",
        "bioinstruction-vicuna15-7b"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": "cardiffnlp/twitter-roberta-base-sentiment-latest fallback",
          "reporting_status": "reported",
          "type": "keyword-first binary extraction with sentiment-model fallback, followed by MCC"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests the task-formatted answer and says not to explain or repeat."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes a user-prompt template and shows no system prompt for open-source models.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-rpi-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-llama31-8b-instruct",
          "n": 4164,
          "notes": "RPI creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 3.82
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-rpi-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-qwen2-7b",
          "n": 4164,
          "notes": "RPI creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -2.15
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-rpi-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-llama2-7b-chat",
          "n": 4164,
          "notes": "RPI creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 5.87
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-rpi-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-alpaca-7b",
          "n": 4164,
          "notes": "RPI creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 4.38
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-rpi-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-glm4-9b-chat",
          "n": 4164,
          "notes": "RPI creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.13
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-rpi-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-vicuna15-7b",
          "n": 4164,
          "notes": "RPI creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-rpi-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-galactica-13b",
          "n": 4164,
          "notes": "RPI creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.24
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-rpi-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-instructprotein-13b",
          "n": 4164,
          "notes": "RPI creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -1.55
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-rpi-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-llama-molinst-protein-7b",
          "n": 4164,
          "notes": "RPI creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 3.71
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-rpi-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-biomedgpt-lm-7b",
          "n": 4164,
          "notes": "RPI creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -2.39
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 4164,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-rpi-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-sirna",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-sirna-closed-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-sirna-closed-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (siRNA test split); Appendix A.3; Table 8; Table 9 closed-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-sirna-closed-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 7 (siRNA column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-sirna-closed-baselines",
      "metrics": [
        {
          "aggregation": "Creator mixed score combining capped MAE and range-MAE-weighted binary F1, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "mixed-score",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Mixed Score",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-gpt4o",
        "bioinstruction-gpt4o-mini"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "first-number extraction followed by the creator mixed-score implementation"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests a direct JSON answer and no chain-of-thought."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes the complete biology-assistant system and user prompt for closed-source models.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-sirna-closed-baselines-results"
          ],
          "metric_id": "mixed-score",
          "model_id": "bioinstruction-gpt4o-mini",
          "n": 6688,
          "notes": "siRNA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 30.37
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-sirna-closed-baselines-results"
          ],
          "metric_id": "mixed-score",
          "model_id": "bioinstruction-gpt4o",
          "n": 6688,
          "notes": "siRNA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 6688,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-sirna-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-sirna",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-sirna-creator-systems-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-sirna-creator-systems-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (siRNA test split); Appendix A.3; Table 8; Section 4.2 Psc prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-sirna-creator-systems-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 7 (siRNA column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-sirna-creator-systems",
      "metrics": [
        {
          "aggregation": "Creator mixed score combining capped MAE and range-MAE-weighted binary F1, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "mixed-score",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Mixed Score",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-chatmultiomics-balanced",
        "bioinstruction-chatmultiomics-stage12",
        "bioinstruction-chatmultiomics-stage123",
        "bioinstruction-chatmultiomics-stage2"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "first-number extraction followed by the creator mixed-score implementation"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "Psc requests clear, concise task answers and numeric output for regression tasks."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Section 4.2 publishes the Psc system prompt used for task performance computation.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-sirna-creator-systems-results"
          ],
          "metric_id": "mixed-score",
          "model_id": "bioinstruction-chatmultiomics-balanced",
          "n": 6688,
          "notes": "siRNA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 42.92
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-sirna-creator-systems-results"
          ],
          "metric_id": "mixed-score",
          "model_id": "bioinstruction-chatmultiomics-stage2",
          "n": 6688,
          "notes": "siRNA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 4.25
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-sirna-creator-systems-results"
          ],
          "metric_id": "mixed-score",
          "model_id": "bioinstruction-chatmultiomics-stage12",
          "n": 6688,
          "notes": "siRNA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 56.31
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-sirna-creator-systems-results"
          ],
          "metric_id": "mixed-score",
          "model_id": "bioinstruction-chatmultiomics-stage123",
          "n": 6688,
          "notes": "siRNA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 56.25
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 6688,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-sirna-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-sirna",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-sirna-open-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-sirna-open-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (siRNA test split); Appendix A.3; Table 8; Table 9 open-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-sirna-open-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 7 (siRNA column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-sirna-open-baselines",
      "metrics": [
        {
          "aggregation": "Creator mixed score combining capped MAE and range-MAE-weighted binary F1, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "mixed-score",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Mixed Score",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-alpaca-7b",
        "bioinstruction-biomedgpt-lm-7b",
        "bioinstruction-galactica-13b",
        "bioinstruction-glm4-9b-chat",
        "bioinstruction-instructprotein-13b",
        "bioinstruction-llama-molinst-protein-7b",
        "bioinstruction-llama2-7b-chat",
        "bioinstruction-llama31-8b-instruct",
        "bioinstruction-qwen2-7b",
        "bioinstruction-vicuna15-7b"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "first-number extraction followed by the creator mixed-score implementation"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests the task-formatted answer and says not to explain or repeat."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes a user-prompt template and shows no system prompt for open-source models.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-sirna-open-baselines-results"
          ],
          "metric_id": "mixed-score",
          "model_id": "bioinstruction-llama31-8b-instruct",
          "n": 6688,
          "notes": "siRNA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 32.76
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-sirna-open-baselines-results"
          ],
          "metric_id": "mixed-score",
          "model_id": "bioinstruction-qwen2-7b",
          "n": 6688,
          "notes": "siRNA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 33.39
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-sirna-open-baselines-results"
          ],
          "metric_id": "mixed-score",
          "model_id": "bioinstruction-llama2-7b-chat",
          "n": 6688,
          "notes": "siRNA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 17.43
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-sirna-open-baselines-results"
          ],
          "metric_id": "mixed-score",
          "model_id": "bioinstruction-alpaca-7b",
          "n": 6688,
          "notes": "siRNA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 19.12
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-sirna-open-baselines-results"
          ],
          "metric_id": "mixed-score",
          "model_id": "bioinstruction-glm4-9b-chat",
          "n": 6688,
          "notes": "siRNA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 23.33
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-sirna-open-baselines-results"
          ],
          "metric_id": "mixed-score",
          "model_id": "bioinstruction-vicuna15-7b",
          "n": 6688,
          "notes": "siRNA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 14.28
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-sirna-open-baselines-results"
          ],
          "metric_id": "mixed-score",
          "model_id": "bioinstruction-galactica-13b",
          "n": 6688,
          "notes": "siRNA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 33.55
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-sirna-open-baselines-results"
          ],
          "metric_id": "mixed-score",
          "model_id": "bioinstruction-instructprotein-13b",
          "n": 6688,
          "notes": "siRNA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 5.58
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-sirna-open-baselines-results"
          ],
          "metric_id": "mixed-score",
          "model_id": "bioinstruction-llama-molinst-protein-7b",
          "n": 6688,
          "notes": "siRNA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 13.85
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-sirna-open-baselines-results"
          ],
          "metric_id": "mixed-score",
          "model_id": "bioinstruction-biomedgpt-lm-7b",
          "n": 6688,
          "notes": "siRNA creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 19.71
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 6688,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-sirna-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-solubility",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-solubility-closed-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-solubility-closed-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (Sol test split); Appendix A.3; Table 8; Table 9 closed-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-solubility-closed-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 6 (Sol column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-solubility-closed-baselines",
      "metrics": [
        {
          "aggregation": "Exact binary accuracy across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Accuracy",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-gpt4o",
        "bioinstruction-gpt4o-mini"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": "cardiffnlp/twitter-roberta-base-sentiment-latest fallback",
          "reporting_status": "reported",
          "type": "keyword-first binary extraction with sentiment-model fallback, followed by accuracy"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests a direct JSON answer and no chain-of-thought."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes the complete biology-assistant system and user prompt for closed-source models.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-solubility-closed-baselines-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bioinstruction-gpt4o-mini",
          "n": 2001,
          "notes": "Sol creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 50.02
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-solubility-closed-baselines-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bioinstruction-gpt4o",
          "n": 2001,
          "notes": "Sol creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 51.67
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 2001,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-solubility-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-solubility",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-solubility-creator-systems-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-solubility-creator-systems-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (Sol test split); Appendix A.3; Table 8; Section 4.2 Psc prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-solubility-creator-systems-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 6 (Sol column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-solubility-creator-systems",
      "metrics": [
        {
          "aggregation": "Exact binary accuracy across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Accuracy",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-chatmultiomics-balanced",
        "bioinstruction-chatmultiomics-stage12",
        "bioinstruction-chatmultiomics-stage123",
        "bioinstruction-chatmultiomics-stage2"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": "cardiffnlp/twitter-roberta-base-sentiment-latest fallback",
          "reporting_status": "reported",
          "type": "keyword-first binary extraction with sentiment-model fallback, followed by accuracy"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "Psc requests clear, concise task answers and numeric output for regression tasks."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Section 4.2 publishes the Psc system prompt used for task performance computation.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-solubility-creator-systems-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bioinstruction-chatmultiomics-balanced",
          "n": 2001,
          "notes": "Sol creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 52.37
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-solubility-creator-systems-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bioinstruction-chatmultiomics-stage2",
          "n": 2001,
          "notes": "Sol creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 49.28
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-solubility-creator-systems-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bioinstruction-chatmultiomics-stage12",
          "n": 2001,
          "notes": "Sol creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 62.07
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-solubility-creator-systems-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bioinstruction-chatmultiomics-stage123",
          "n": 2001,
          "notes": "Sol creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 63.02
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 2001,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-solubility-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-solubility",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-solubility-open-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-solubility-open-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (Sol test split); Appendix A.3; Table 8; Table 9 open-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-solubility-open-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 6 (Sol column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-solubility-open-baselines",
      "metrics": [
        {
          "aggregation": "Exact binary accuracy across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Accuracy",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-alpaca-7b",
        "bioinstruction-biomedgpt-lm-7b",
        "bioinstruction-galactica-13b",
        "bioinstruction-glm4-9b-chat",
        "bioinstruction-instructprotein-13b",
        "bioinstruction-llama-molinst-protein-7b",
        "bioinstruction-llama2-7b-chat",
        "bioinstruction-llama31-8b-instruct",
        "bioinstruction-qwen2-7b",
        "bioinstruction-vicuna15-7b"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": "cardiffnlp/twitter-roberta-base-sentiment-latest fallback",
          "reporting_status": "reported",
          "type": "keyword-first binary extraction with sentiment-model fallback, followed by accuracy"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests the task-formatted answer and says not to explain or repeat."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes a user-prompt template and shows no system prompt for open-source models.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-solubility-open-baselines-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bioinstruction-llama31-8b-instruct",
          "n": 2001,
          "notes": "Sol creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 50.27
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-solubility-open-baselines-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bioinstruction-qwen2-7b",
          "n": 2001,
          "notes": "Sol creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 52.52
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-solubility-open-baselines-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bioinstruction-llama2-7b-chat",
          "n": 2001,
          "notes": "Sol creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 49.48
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-solubility-open-baselines-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bioinstruction-alpaca-7b",
          "n": 2001,
          "notes": "Sol creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 50.12
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-solubility-open-baselines-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bioinstruction-glm4-9b-chat",
          "n": 2001,
          "notes": "Sol creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 50.72
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-solubility-open-baselines-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bioinstruction-vicuna15-7b",
          "n": 2001,
          "notes": "Sol creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 51.57
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-solubility-open-baselines-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bioinstruction-galactica-13b",
          "n": 2001,
          "notes": "Sol creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 46.78
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-solubility-open-baselines-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bioinstruction-instructprotein-13b",
          "n": 2001,
          "notes": "Sol creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 47.88
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-solubility-open-baselines-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bioinstruction-llama-molinst-protein-7b",
          "n": 2001,
          "notes": "Sol creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 48.33
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-solubility-open-baselines-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bioinstruction-biomedgpt-lm-7b",
          "n": 2001,
          "notes": "Sol creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 49.78
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 2001,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-solubility-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-stability",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-stability-closed-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-stability-closed-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (Sta test split); Appendix A.3; Table 8; Table 9 closed-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-stability-closed-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 6 (Sta column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-stability-closed-baselines",
      "metrics": [
        {
          "aggregation": "Spearman rank correlation across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "spearman-rho",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "Spearman's ρ",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-gpt4o",
        "bioinstruction-gpt4o-mini"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "first-number extraction followed by Spearman rank correlation"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests a direct JSON answer and no chain-of-thought."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes the complete biology-assistant system and user prompt for closed-source models.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-stability-closed-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-gpt4o-mini",
          "n": 12851,
          "notes": "Sta creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -1.52
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-stability-closed-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-gpt4o",
          "n": 12851,
          "notes": "Sta creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.09
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 12851,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-stability-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-stability",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-stability-creator-systems-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-stability-creator-systems-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (Sta test split); Appendix A.3; Table 8; Section 4.2 Psc prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-stability-creator-systems-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 6 (Sta column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-stability-creator-systems",
      "metrics": [
        {
          "aggregation": "Spearman rank correlation across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "spearman-rho",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "Spearman's ρ",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-chatmultiomics-balanced",
        "bioinstruction-chatmultiomics-stage12",
        "bioinstruction-chatmultiomics-stage123",
        "bioinstruction-chatmultiomics-stage2"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "first-number extraction followed by Spearman rank correlation"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "Psc requests clear, concise task answers and numeric output for regression tasks."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Section 4.2 publishes the Psc system prompt used for task performance computation.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-stability-creator-systems-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-chatmultiomics-balanced",
          "n": 12851,
          "notes": "Sta creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.48
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-stability-creator-systems-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-chatmultiomics-stage2",
          "n": 12851,
          "notes": "Sta creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.23
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-stability-creator-systems-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-chatmultiomics-stage12",
          "n": 12851,
          "notes": "Sta creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 56.76
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-stability-creator-systems-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-chatmultiomics-stage123",
          "n": 12851,
          "notes": "Sta creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 60.25
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 12851,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-stability-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-stability",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-stability-open-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-stability-open-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (Sta test split); Appendix A.3; Table 8; Table 9 open-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-stability-open-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 6 (Sta column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-stability-open-baselines",
      "metrics": [
        {
          "aggregation": "Spearman rank correlation across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "spearman-rho",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "Spearman's ρ",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-alpaca-7b",
        "bioinstruction-biomedgpt-lm-7b",
        "bioinstruction-galactica-13b",
        "bioinstruction-glm4-9b-chat",
        "bioinstruction-instructprotein-13b",
        "bioinstruction-llama-molinst-protein-7b",
        "bioinstruction-llama2-7b-chat",
        "bioinstruction-llama31-8b-instruct",
        "bioinstruction-qwen2-7b",
        "bioinstruction-vicuna15-7b"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "first-number extraction followed by Spearman rank correlation"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests the task-formatted answer and says not to explain or repeat."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes a user-prompt template and shows no system prompt for open-source models.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-stability-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-llama31-8b-instruct",
          "n": 12851,
          "notes": "Sta creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.61
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-stability-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-qwen2-7b",
          "n": 12851,
          "notes": "Sta creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -5.86
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-stability-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-llama2-7b-chat",
          "n": 12851,
          "notes": "Sta creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.51
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-stability-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-alpaca-7b",
          "n": 12851,
          "notes": "Sta creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 2.05
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-stability-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-glm4-9b-chat",
          "n": 12851,
          "notes": "Sta creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -2.72
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-stability-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-vicuna15-7b",
          "n": 12851,
          "notes": "Sta creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 5.65
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-stability-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-galactica-13b",
          "n": 12851,
          "notes": "Sta creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.52
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-stability-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-instructprotein-13b",
          "n": 12851,
          "notes": "Sta creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.35
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-stability-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-llama-molinst-protein-7b",
          "n": 12851,
          "notes": "Sta creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.05
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-stability-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-biomedgpt-lm-7b",
          "n": 12851,
          "notes": "Sta creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.92
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 12851,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-stability-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-tb-human",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-tb-human-closed-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-tb-human-closed-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (TB-H test split); Appendix A.3; Table 8; Table 9 closed-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-tb-human-closed-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 4 (TB-H column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-tb-human-closed-baselines",
      "metrics": [
        {
          "aggregation": "Matthews correlation coefficient across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "mcc",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "MCC",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-gpt4o",
        "bioinstruction-gpt4o-mini"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": "cardiffnlp/twitter-roberta-base-sentiment-latest fallback",
          "reporting_status": "reported",
          "type": "keyword-first binary extraction with sentiment-model fallback, followed by MCC"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests a direct JSON answer and no chain-of-thought."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes the complete biology-assistant system and user prompt for closed-source models.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-tb-human-closed-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-gpt4o-mini",
          "n": 5000,
          "notes": "TB-H creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.14
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-tb-human-closed-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-gpt4o",
          "n": 5000,
          "notes": "TB-H creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -1.7
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 5000,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-tb-human-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-tb-human",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-tb-human-creator-systems-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-tb-human-creator-systems-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (TB-H test split); Appendix A.3; Table 8; Section 4.2 Psc prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-tb-human-creator-systems-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 4 (TB-H column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-tb-human-creator-systems",
      "metrics": [
        {
          "aggregation": "Matthews correlation coefficient across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "mcc",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "MCC",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-chatmultiomics-balanced",
        "bioinstruction-chatmultiomics-stage12",
        "bioinstruction-chatmultiomics-stage123",
        "bioinstruction-chatmultiomics-stage2"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": "cardiffnlp/twitter-roberta-base-sentiment-latest fallback",
          "reporting_status": "reported",
          "type": "keyword-first binary extraction with sentiment-model fallback, followed by MCC"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "Psc requests clear, concise task answers and numeric output for regression tasks."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Section 4.2 publishes the Psc system prompt used for task performance computation.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-tb-human-creator-systems-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-chatmultiomics-balanced",
          "n": 5000,
          "notes": "TB-H creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 2.46
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-tb-human-creator-systems-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-chatmultiomics-stage2",
          "n": 5000,
          "notes": "TB-H creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.86
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-tb-human-creator-systems-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-chatmultiomics-stage12",
          "n": 5000,
          "notes": "TB-H creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 19.07
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-tb-human-creator-systems-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-chatmultiomics-stage123",
          "n": 5000,
          "notes": "TB-H creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 24.45
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 5000,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-tb-human-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-tb-human",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-tb-human-open-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-tb-human-open-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (TB-H test split); Appendix A.3; Table 8; Table 9 open-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-tb-human-open-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 4 (TB-H column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-tb-human-open-baselines",
      "metrics": [
        {
          "aggregation": "Matthews correlation coefficient across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "mcc",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "MCC",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-alpaca-7b",
        "bioinstruction-galactica-13b",
        "bioinstruction-glm4-9b-chat",
        "bioinstruction-instructprotein-13b",
        "bioinstruction-llama-molinst-protein-7b",
        "bioinstruction-llama2-7b-chat",
        "bioinstruction-llama31-8b-instruct",
        "bioinstruction-qwen2-7b",
        "bioinstruction-vicuna15-7b"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": "cardiffnlp/twitter-roberta-base-sentiment-latest fallback",
          "reporting_status": "reported",
          "type": "keyword-first binary extraction with sentiment-model fallback, followed by MCC"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests the task-formatted answer and says not to explain or repeat."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes a user-prompt template and shows no system prompt for open-source models.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-tb-human-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-llama31-8b-instruct",
          "n": 5000,
          "notes": "TB-H creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-tb-human-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-qwen2-7b",
          "n": 5000,
          "notes": "TB-H creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.21
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-tb-human-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-llama2-7b-chat",
          "n": 5000,
          "notes": "TB-H creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 1.84
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-tb-human-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-alpaca-7b",
          "n": 5000,
          "notes": "TB-H creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 2
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-tb-human-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-glm4-9b-chat",
          "n": 5000,
          "notes": "TB-H creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-tb-human-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-vicuna15-7b",
          "n": 5000,
          "notes": "TB-H creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-tb-human-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-galactica-13b",
          "n": 5000,
          "notes": "TB-H creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 3
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-tb-human-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-instructprotein-13b",
          "n": 5000,
          "notes": "TB-H creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -1.29
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-tb-human-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-llama-molinst-protein-7b",
          "n": 5000,
          "notes": "TB-H creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 2.4
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 5000,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-tb-human-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-tb-mouse",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-tb-mouse-closed-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-tb-mouse-closed-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (TB-M test split); Appendix A.3; Table 8; Table 9 closed-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-tb-mouse-closed-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 4 (TB-M column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-tb-mouse-closed-baselines",
      "metrics": [
        {
          "aggregation": "Matthews correlation coefficient across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "mcc",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "MCC",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-gpt4o",
        "bioinstruction-gpt4o-mini"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": "cardiffnlp/twitter-roberta-base-sentiment-latest fallback",
          "reporting_status": "reported",
          "type": "keyword-first binary extraction with sentiment-model fallback, followed by MCC"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests a direct JSON answer and no chain-of-thought."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes the complete biology-assistant system and user prompt for closed-source models.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-tb-mouse-closed-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-gpt4o-mini",
          "n": 10005,
          "notes": "TB-M creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.31
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-tb-mouse-closed-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-gpt4o",
          "n": 10005,
          "notes": "TB-M creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -1.38
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 10005,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-tb-mouse-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-tb-mouse",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-tb-mouse-creator-systems-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-tb-mouse-creator-systems-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (TB-M test split); Appendix A.3; Table 8; Section 4.2 Psc prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-tb-mouse-creator-systems-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 4 (TB-M column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-tb-mouse-creator-systems",
      "metrics": [
        {
          "aggregation": "Matthews correlation coefficient across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "mcc",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "MCC",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-chatmultiomics-balanced",
        "bioinstruction-chatmultiomics-stage12",
        "bioinstruction-chatmultiomics-stage123",
        "bioinstruction-chatmultiomics-stage2"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": "cardiffnlp/twitter-roberta-base-sentiment-latest fallback",
          "reporting_status": "reported",
          "type": "keyword-first binary extraction with sentiment-model fallback, followed by MCC"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "Psc requests clear, concise task answers and numeric output for regression tasks."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Section 4.2 publishes the Psc system prompt used for task performance computation.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-tb-mouse-creator-systems-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-chatmultiomics-balanced",
          "n": 10005,
          "notes": "TB-M creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.88
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-tb-mouse-creator-systems-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-chatmultiomics-stage2",
          "n": 10005,
          "notes": "TB-M creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.13
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-tb-mouse-creator-systems-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-chatmultiomics-stage12",
          "n": 10005,
          "notes": "TB-M creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 27.94
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-tb-mouse-creator-systems-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-chatmultiomics-stage123",
          "n": 10005,
          "notes": "TB-M creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 39.91
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 10005,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-tb-mouse-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-tb-mouse",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-tb-mouse-open-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-tb-mouse-open-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (TB-M test split); Appendix A.3; Table 8; Table 9 open-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-tb-mouse-open-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 4 (TB-M column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-tb-mouse-open-baselines",
      "metrics": [
        {
          "aggregation": "Matthews correlation coefficient across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "mcc",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "MCC",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-alpaca-7b",
        "bioinstruction-galactica-13b",
        "bioinstruction-glm4-9b-chat",
        "bioinstruction-instructprotein-13b",
        "bioinstruction-llama-molinst-protein-7b",
        "bioinstruction-llama2-7b-chat",
        "bioinstruction-llama31-8b-instruct",
        "bioinstruction-qwen2-7b",
        "bioinstruction-vicuna15-7b"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": "cardiffnlp/twitter-roberta-base-sentiment-latest fallback",
          "reporting_status": "reported",
          "type": "keyword-first binary extraction with sentiment-model fallback, followed by MCC"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests the task-formatted answer and says not to explain or repeat."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes a user-prompt template and shows no system prompt for open-source models.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-tb-mouse-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-llama31-8b-instruct",
          "n": 10005,
          "notes": "TB-M creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -1.42
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-tb-mouse-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-qwen2-7b",
          "n": 10005,
          "notes": "TB-M creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -1.59
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-tb-mouse-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-llama2-7b-chat",
          "n": 10005,
          "notes": "TB-M creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.97
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-tb-mouse-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-alpaca-7b",
          "n": 10005,
          "notes": "TB-M creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-tb-mouse-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-glm4-9b-chat",
          "n": 10005,
          "notes": "TB-M creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-tb-mouse-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-vicuna15-7b",
          "n": 10005,
          "notes": "TB-M creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-tb-mouse-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-galactica-13b",
          "n": 10005,
          "notes": "TB-M creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -2.81
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-tb-mouse-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-instructprotein-13b",
          "n": 10005,
          "notes": "TB-M creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 1.19
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-tb-mouse-open-baselines-results"
          ],
          "metric_id": "mcc",
          "model_id": "bioinstruction-llama-molinst-protein-7b",
          "n": 10005,
          "notes": "TB-M creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.33
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 10005,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-tb-mouse-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-thermostability",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-thermostability-closed-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-thermostability-closed-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (Ther test split); Appendix A.3; Table 8; Table 9 closed-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-thermostability-closed-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 6 (Ther column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-thermostability-closed-baselines",
      "metrics": [
        {
          "aggregation": "Spearman rank correlation across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "spearman-rho",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "Spearman's ρ",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-gpt4o",
        "bioinstruction-gpt4o-mini"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "first-number extraction followed by Spearman rank correlation"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests a direct JSON answer and no chain-of-thought."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes the complete biology-assistant system and user prompt for closed-source models.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-thermostability-closed-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-gpt4o-mini",
          "n": 1336,
          "notes": "Ther creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.32
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-thermostability-closed-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-gpt4o",
          "n": 1336,
          "notes": "Ther creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 3.5
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 1336,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-thermostability-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-thermostability",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-thermostability-creator-systems-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-thermostability-creator-systems-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (Ther test split); Appendix A.3; Table 8; Section 4.2 Psc prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-thermostability-creator-systems-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 6 (Ther column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-thermostability-creator-systems",
      "metrics": [
        {
          "aggregation": "Spearman rank correlation across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "spearman-rho",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "Spearman's ρ",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-chatmultiomics-balanced",
        "bioinstruction-chatmultiomics-stage12",
        "bioinstruction-chatmultiomics-stage123",
        "bioinstruction-chatmultiomics-stage2"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "first-number extraction followed by Spearman rank correlation"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "Psc requests clear, concise task answers and numeric output for regression tasks."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Section 4.2 publishes the Psc system prompt used for task performance computation.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-thermostability-creator-systems-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-chatmultiomics-balanced",
          "n": 1336,
          "notes": "Ther creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 39.97
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-thermostability-creator-systems-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-chatmultiomics-stage2",
          "n": 1336,
          "notes": "Ther creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.51
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-thermostability-creator-systems-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-chatmultiomics-stage12",
          "n": 1336,
          "notes": "Ther creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 44.59
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-thermostability-creator-systems-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-chatmultiomics-stage123",
          "n": 1336,
          "notes": "Ther creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 45.07
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 1336,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-thermostability-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "bioinstruction-thermostability",
      "benchmark_version": "emnlp-2025",
      "comparability_group": "bioinstruction-thermostability-open-baselines-emnlp-2025",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-thermostability-open-baselines-scope-protocol",
          "locator": {
            "note": "Scope, prompt, output parser, grader, scaling, and aggregation.",
            "type": "table",
            "value": "Table 2 (Ther test split); Appendix A.3; Table 8; Table 9 open-source prompt"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bioinstruction-thermostability-open-baselines-results",
          "locator": {
            "note": "All registered model values; literature-SOTA row omitted.",
            "type": "table",
            "value": "Table 6 (Ther column)"
          },
          "source_id": "bioinstructions-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bioinstruction-thermostability-open-baselines",
      "metrics": [
        {
          "aggregation": "Spearman rank correlation across held-out test examples, scaled by 100",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "spearman-rho",
          "pass_threshold": null,
          "range": [
            -100,
            100
          ],
          "source_label": "Spearman's ρ",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bioinstruction-alpaca-7b",
        "bioinstruction-biomedgpt-lm-7b",
        "bioinstruction-galactica-13b",
        "bioinstruction-glm4-9b-chat",
        "bioinstruction-instructprotein-13b",
        "bioinstruction-llama-molinst-protein-7b",
        "bioinstruction-llama2-7b-chat",
        "bioinstruction-llama31-8b-instruct",
        "bioinstruction-qwen2-7b",
        "bioinstruction-vicuna15-7b"
      ],
      "protocol": {
        "contamination": {
          "notes": "No suite-level decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "first-number extraction followed by Spearman rank correlation"
        },
        "reasoning": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The prompt requests the task-formatted answer and says not to explain or repeat."
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published prompt contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Metrics are never averaged across tasks.",
          "reporting_status": "reported",
          "value": "Point metric over the complete published test split, scaled by 100 and rounded to two decimals; no confidence interval is reported."
        },
        "system_prompt_public": {
          "notes": "Table 9 publishes a user-prompt template and shows no system prompt for open-source models.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No common evaluation container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Direct model inference over supplied sequence prompts; no agent tool calls are part of the protocol.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One sequence-conditioned prompt produces one answer.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-thermostability-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-llama31-8b-instruct",
          "n": 1336,
          "notes": "Ther creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 4.67
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-thermostability-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-qwen2-7b",
          "n": 1336,
          "notes": "Ther creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.93
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-thermostability-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-llama2-7b-chat",
          "n": 1336,
          "notes": "Ther creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.4
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-thermostability-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-alpaca-7b",
          "n": 1336,
          "notes": "Ther creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 2.27
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-thermostability-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-glm4-9b-chat",
          "n": 1336,
          "notes": "Ther creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 1.4
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-thermostability-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-vicuna15-7b",
          "n": 1336,
          "notes": "Ther creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 0.9
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-thermostability-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-galactica-13b",
          "n": 1336,
          "notes": "Ther creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.58
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-thermostability-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-instructprotein-13b",
          "n": 1336,
          "notes": "Ther creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.5
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-thermostability-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-llama-molinst-protein-7b",
          "n": 1336,
          "notes": "Ther creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": 1.07
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bioinstruction-thermostability-open-baselines-results"
          ],
          "metric_id": "spearman-rho",
          "model_id": "bioinstruction-biomedgpt-lm-7b",
          "n": 1336,
          "notes": "Ther creator-paper result; metric scaled by 100.",
          "status": "verified",
          "value": -0.72
        }
      ],
      "scope": {
        "filter": "Published held-out test split in Table 2.",
        "n": 1336,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "bioinstruction-thermostability-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Prompt-distinct model groups are separate runs. Literature-SOTA rows are excluded because they do not use this evaluation prompt.",
        "status": "verified"
      },
      "work_id": "bioinstructions-paper",
      "work_version_id": "bioinstructions-paper-2025-11-04"
    },
    {
      "benchmark_id": "biomysterybench",
      "benchmark_version": "v8",
      "comparability_group": "biomysterybench-v8-full-five-episodes-protocol",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "biomysterybench-evidence-run-scope",
          "locator": {
            "note": "Reports 99 problems partitioned into 76 and 23 after four failed-QC candidates were removed.",
            "type": "section",
            "value": "“Benchmarking models on verifiable biological tasks,” “Human-solvable,” and “Human-difficult”"
          },
          "source_id": "biomysterybench-official",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/scope"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "biomysterybench-evidence-run-protocol",
          "locator": {
            "note": "Reports containers, databases, package installation, final-answer grading, five episodes, bootstrap-within-problem error bars, and solve-count reliability profiles.",
            "type": "section",
            "value": "Environment description, method-agnostic property, Figures 1–3 captions, and continuing reliability analysis"
          },
          "source_id": "biomysterybench-official",
          "source_type": "work",
          "supports": [
            "/protocol/turns",
            "/protocol/tools/internet",
            "/protocol/tools/databases",
            "/protocol/tools/code_execution",
            "/protocol/tools/container",
            "/protocol/tools/external_tools",
            "/protocol/repeats",
            "/protocol/grader",
            "/protocol/statistical",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "biomysterybench-evidence-run-leak-control",
          "locator": {
            "note": "Documents prohibited accession/reverse lookup, permitted standard database use, and 16 scrubbed v8 problems.",
            "type": "repository-path",
            "value": "README.md Rules and CHANGELOG.md v8 entry at commit 51c9024021b8989a0cb06ae623b02f90d14c2da3"
          },
          "source_id": "biomysterybench-preview-resource",
          "source_type": "resource",
          "supports": [
            "/protocol/contamination"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "biomysterybench-evidence-run-unreported",
          "locator": {
            "note": "The public protocol does not report shots, a browser, a system prompt, effort, budgets, temperature, seed, human-review status, exact grader implementation, bootstrap count, or numeric confidence bounds.",
            "type": "section",
            "value": "Complete public report and figure captions"
          },
          "source_id": "biomysterybench-official",
          "source_type": "work",
          "supports": [
            "/protocol/shots",
            "/protocol/system_prompt_public",
            "/protocol/reasoning",
            "/protocol/tools/browser",
            "/protocol/token_budget",
            "/protocol/time_budget",
            "/protocol/temperature",
            "/protocol/seed"
          ]
        }
      ],
      "id": "biomysterybench-official-run",
      "metrics": [
        {
          "aggregation": "mean over five trials per problem, then mean over the selected problems",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "episode-accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Accuracy",
          "tolerance": "expert-authored answer rubric",
          "unit": "percent"
        }
      ],
      "model_ids": [
        "claude-haiku-4-5",
        "claude-mythos-preview",
        "claude-opus-4-6",
        "claude-opus-4-7",
        "claude-sonnet-4-6"
      ],
      "protocol": {
        "contamination": {
          "notes": "This is an anti-cheating/data-leak control, not a reported pretraining-contamination analysis.",
          "reporting_status": "reported",
          "value": "source-dataset reverse identification prohibited; accession-ID leaks scrubbed from 16 v8 problems"
        },
        "grader": {
          "human_review": null,
          "model": null,
          "reporting_status": "reported",
          "type": "objective final-answer scoring against an expert-authored answer rubric"
        },
        "reasoning": {
          "notes": "Model-specific reasoning or effort settings are not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "repeats": {
          "notes": "Each model attempted every problem in five independent episodes.",
          "reporting_status": "reported",
          "value": 5
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The report does not characterize the agent setup using a zero-shot or few-shot convention.",
          "reporting_status": "not_reported",
          "value": null
        },
        "statistical": {
          "notes": "The report does not publish numeric confidence bounds, the number of bootstrap resamples, or the exact grader implementation.",
          "reporting_status": "reported",
          "value": "accuracy averaged over five episodes per problem with error bars from bootstrap sampling within problems; per-problem reliability analyzed by 0-of-5 through 5-of-5 solve counts"
        },
        "system_prompt_public": {
          "notes": "Problem questions and v11 answer rubrics are released, but the report does not publish or state the availability of the evaluation system prompt.",
          "reporting_status": "not_reported",
          "value": null
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "A browser interface is not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "code_execution": {
            "notes": "The environment supports analysis code and command-line bioinformatics workflows.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "Every model episode ran in a container initialized with a minimal canonical tool set.",
            "reporting_status": "reported",
            "value": true
          },
          "databases": {
            "notes": "Standard database uses such as gene lookup, sequence annotation, reference-genome download, and BLAST were permitted.",
            "reporting_status": "reported",
            "value": [
              "NCBI",
              "Ensembl",
              "other canonical bioinformatics databases allowed per problem"
            ]
          },
          "external_tools": {
            "notes": "The exact base image and complete package inventory are not published.",
            "reporting_status": "reported",
            "value": "preinstalled canonical bioinformatics tools plus packages installable through pip and conda"
          },
          "internet": {
            "notes": "Agents could access canonical bioinformatics databases and download references; reverse-identifying source datasets through accession lookup was prohibited.",
            "reporting_status": "reported",
            "value": "allowlisted external access"
          }
        },
        "turns": {
          "notes": "Each run is an open-ended tool-using trajectory ending in a final answer.",
          "reporting_status": "reported",
          "value": "multi-turn agent episode"
        }
      },
      "results": [],
      "scope": {
        "filter": "Initial v8 release used by the April report; four failed-QC human-difficult candidates had already been excluded before the 99-problem benchmark was formed.",
        "n": 99,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Full v8 scope and common protocol are retained here; numeric results are split into separate 76- and 23-problem runs because the report publishes those subsets separately.",
        "status": "verified"
      },
      "work_id": "biomysterybench-official",
      "work_version_id": "biomysterybench-official-2026-04-29"
    },
    {
      "benchmark_id": "biomysterybench",
      "benchmark_version": "v8",
      "comparability_group": "biomysterybench-v8-human-difficult-five-episodes",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "biomysterybench-evidence-difficult-scope",
          "locator": {
            "note": "Defines the 23-problem subset, five episodes, and within-problem bootstrap.",
            "type": "section",
            "value": "“Human-difficult” and Figure 2 caption"
          },
          "source_id": "biomysterybench-official",
          "source_type": "work",
          "supports": [
            "/scope",
            "/protocol/repeats",
            "/protocol/statistical",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "biomysterybench-evidence-difficult-results",
          "locator": {
            "note": "Printed labels report 5.2, 19.1, 23.5, 27.0, and 29.6 percent.",
            "type": "figure",
            "value": "Figure 2: BioMysteryBench human-difficult set (23 problems)"
          },
          "source_id": "biomysterybench-figure-two-resource",
          "source_type": "resource",
          "supports": [
            "/results"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "biomysterybench-evidence-difficult-protocol",
          "locator": {
            "note": "Reports the agent trajectory, container, databases, code/tool access, package installation, and final-answer rubric grading.",
            "type": "section",
            "value": "Environment description and method-agnostic property"
          },
          "source_id": "biomysterybench-official",
          "source_type": "work",
          "supports": [
            "/protocol/turns",
            "/protocol/tools/internet",
            "/protocol/tools/databases",
            "/protocol/tools/code_execution",
            "/protocol/tools/container",
            "/protocol/tools/external_tools",
            "/protocol/grader"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "biomysterybench-evidence-difficult-leak-control",
          "locator": {
            "note": "Documents prohibited reverse lookup, permitted standard database use, and 16 scrubbed v8 problems.",
            "type": "repository-path",
            "value": "README.md Rules and CHANGELOG.md v8 entry at commit 51c9024021b8989a0cb06ae623b02f90d14c2da3"
          },
          "source_id": "biomysterybench-preview-resource",
          "source_type": "resource",
          "supports": [
            "/protocol/contamination"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "biomysterybench-evidence-difficult-unreported",
          "locator": {
            "note": "Does not report shots, browser interface, system prompt, effort, budgets, temperature, seed, human-review status, bootstrap count, or numeric confidence bounds.",
            "type": "section",
            "value": "Complete public report and Figure 2 caption"
          },
          "source_id": "biomysterybench-official",
          "source_type": "work",
          "supports": [
            "/protocol/shots",
            "/protocol/system_prompt_public",
            "/protocol/reasoning",
            "/protocol/tools/browser",
            "/protocol/token_budget",
            "/protocol/time_budget",
            "/protocol/temperature",
            "/protocol/seed"
          ]
        }
      ],
      "id": "biomysterybench-v8-human-difficult",
      "metrics": [
        {
          "aggregation": "mean over five episodes per problem, then mean over 23 problems",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "episode-accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Accuracy — human-difficult",
          "tolerance": "expert-authored answer rubric",
          "unit": "percent"
        }
      ],
      "model_ids": [
        "claude-haiku-4-5",
        "claude-mythos-preview",
        "claude-opus-4-6",
        "claude-opus-4-7",
        "claude-sonnet-4-6"
      ],
      "protocol": {
        "contamination": {
          "notes": "Anti-cheating control; no pretraining-contamination analysis is reported.",
          "reporting_status": "reported",
          "value": "source-dataset reverse identification prohibited; accession-ID leaks scrubbed from 16 v8 problems"
        },
        "grader": {
          "human_review": null,
          "model": null,
          "reporting_status": "reported",
          "type": "objective final-answer scoring against an expert-authored answer rubric"
        },
        "reasoning": {
          "notes": "Model-specific reasoning or effort settings are not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "repeats": {
          "notes": "Five independent episodes per model and problem.",
          "reporting_status": "reported",
          "value": 5
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "statistical": {
          "notes": "Numeric confidence bounds and bootstrap-resample count are not published.",
          "reporting_status": "reported",
          "value": "accuracy averaged over five episodes per problem with error bars from bootstrap sampling within problems; per-problem reliability also reported by 0-of-5 through 5-of-5 solve counts"
        },
        "system_prompt_public": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "A browser interface is not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "code_execution": {
            "notes": "Analysis code and command-line workflows were supported.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "Each episode used a container with canonical bioinformatics tools.",
            "reporting_status": "reported",
            "value": true
          },
          "databases": {
            "notes": "Reference downloads and standard bioinformatics queries were permitted.",
            "reporting_status": "reported",
            "value": "canonical bioinformatics databases including NCBI and Ensembl"
          },
          "external_tools": {
            "notes": "Exact inventory not reported.",
            "reporting_status": "reported",
            "value": "preinstalled tools plus packages installable through pip and conda"
          },
          "internet": {
            "notes": "Canonical bioinformatics resources were allowed; source-dataset reverse lookup was prohibited.",
            "reporting_status": "reported",
            "value": "allowlisted external access"
          }
        },
        "turns": {
          "notes": "Open-ended tool-using trajectory ending in a final answer.",
          "reporting_status": "reported",
          "value": "multi-turn agent episode"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "biomysterybench-evidence-difficult-results"
          ],
          "metric_id": "episode-accuracy",
          "model_id": "claude-haiku-4-5",
          "n": 23,
          "notes": "115 model episodes; official figure labels the point estimate and draws unlabeled bootstrap error bars.",
          "status": "verified",
          "value": 5.2
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "biomysterybench-evidence-difficult-results"
          ],
          "metric_id": "episode-accuracy",
          "model_id": "claude-sonnet-4-6",
          "n": 23,
          "notes": "115 model episodes; official figure labels the point estimate and draws unlabeled bootstrap error bars.",
          "status": "verified",
          "value": 19.1
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "biomysterybench-evidence-difficult-results"
          ],
          "metric_id": "episode-accuracy",
          "model_id": "claude-opus-4-6",
          "n": 23,
          "notes": "115 model episodes; official figure labels the point estimate and draws unlabeled bootstrap error bars.",
          "status": "verified",
          "value": 23.5
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "biomysterybench-evidence-difficult-results"
          ],
          "metric_id": "episode-accuracy",
          "model_id": "claude-opus-4-7",
          "n": 23,
          "notes": "115 model episodes; official figure labels the point estimate and draws unlabeled bootstrap error bars.",
          "status": "verified",
          "value": 27.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "biomysterybench-evidence-difficult-results"
          ],
          "metric_id": "episode-accuracy",
          "model_id": "claude-mythos-preview",
          "n": 23,
          "notes": "115 model episodes; official figure labels the point estimate and draws unlabeled bootstrap error bars.",
          "status": "verified",
          "value": 29.6
        }
      ],
      "scope": {
        "filter": "Problems not solved by the human panel after four failed-QC candidates were removed.",
        "n": 23,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "human-difficult",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "All five printed Figure 2 labels transcribed from the immutable official image; error-bar bounds are intentionally left null.",
        "status": "verified"
      },
      "work_id": "biomysterybench-official",
      "work_version_id": "biomysterybench-official-2026-04-29"
    },
    {
      "benchmark_id": "biomysterybench",
      "benchmark_version": "v8",
      "comparability_group": "biomysterybench-v8-human-solvable-five-episodes",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "biomysterybench-evidence-solvable-scope",
          "locator": {
            "note": "Defines the 76-problem subset, five trials, and within-problem bootstrap.",
            "type": "section",
            "value": "“Human-solvable” and Figure 1 caption"
          },
          "source_id": "biomysterybench-official",
          "source_type": "work",
          "supports": [
            "/scope",
            "/protocol/repeats",
            "/protocol/statistical",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "biomysterybench-evidence-solvable-results",
          "locator": {
            "note": "Printed labels report 36.8, 71.8, 77.4, 78.9, and 82.6 percent.",
            "type": "figure",
            "value": "Figure 1: BioMysteryBench human-solvable set (76 problems)"
          },
          "source_id": "biomysterybench-figure-one-resource",
          "source_type": "resource",
          "supports": [
            "/results"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "biomysterybench-evidence-solvable-protocol",
          "locator": {
            "note": "Reports the agent trajectory, container, databases, code/tool access, package installation, and final-answer rubric grading.",
            "type": "section",
            "value": "Environment description and method-agnostic property"
          },
          "source_id": "biomysterybench-official",
          "source_type": "work",
          "supports": [
            "/protocol/turns",
            "/protocol/tools/internet",
            "/protocol/tools/databases",
            "/protocol/tools/code_execution",
            "/protocol/tools/container",
            "/protocol/tools/external_tools",
            "/protocol/grader"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "biomysterybench-evidence-solvable-leak-control",
          "locator": {
            "note": "Documents prohibited reverse lookup, permitted standard database use, and 16 scrubbed v8 problems.",
            "type": "repository-path",
            "value": "README.md Rules and CHANGELOG.md v8 entry at commit 51c9024021b8989a0cb06ae623b02f90d14c2da3"
          },
          "source_id": "biomysterybench-preview-resource",
          "source_type": "resource",
          "supports": [
            "/protocol/contamination"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "biomysterybench-evidence-solvable-unreported",
          "locator": {
            "note": "Does not report shots, browser interface, system prompt, effort, budgets, temperature, seed, human-review status, bootstrap count, or numeric confidence bounds.",
            "type": "section",
            "value": "Complete public report and Figure 1 caption"
          },
          "source_id": "biomysterybench-official",
          "source_type": "work",
          "supports": [
            "/protocol/shots",
            "/protocol/system_prompt_public",
            "/protocol/reasoning",
            "/protocol/tools/browser",
            "/protocol/token_budget",
            "/protocol/time_budget",
            "/protocol/temperature",
            "/protocol/seed"
          ]
        }
      ],
      "id": "biomysterybench-v8-human-solvable",
      "metrics": [
        {
          "aggregation": "mean over five trials per problem, then mean over 76 problems",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "episode-accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Accuracy — human-solvable",
          "tolerance": "expert-authored answer rubric",
          "unit": "percent"
        }
      ],
      "model_ids": [
        "claude-haiku-4-5",
        "claude-mythos-preview",
        "claude-opus-4-6",
        "claude-opus-4-7",
        "claude-sonnet-4-6"
      ],
      "protocol": {
        "contamination": {
          "notes": "Anti-cheating control; no pretraining-contamination analysis is reported.",
          "reporting_status": "reported",
          "value": "source-dataset reverse identification prohibited; accession-ID leaks scrubbed from 16 v8 problems"
        },
        "grader": {
          "human_review": null,
          "model": null,
          "reporting_status": "reported",
          "type": "objective final-answer scoring against an expert-authored answer rubric"
        },
        "reasoning": {
          "notes": "Model-specific reasoning or effort settings are not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "repeats": {
          "notes": "Five independent episodes per model and problem.",
          "reporting_status": "reported",
          "value": 5
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "statistical": {
          "notes": "Numeric confidence bounds and bootstrap-resample count are not published.",
          "reporting_status": "reported",
          "value": "accuracy averaged over five trials per problem with error bars from bootstrap sampling within problems; per-problem reliability also reported by 0-of-5 through 5-of-5 solve counts"
        },
        "system_prompt_public": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "A browser interface is not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "code_execution": {
            "notes": "Analysis code and command-line workflows were supported.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "Each episode used a container with canonical bioinformatics tools.",
            "reporting_status": "reported",
            "value": true
          },
          "databases": {
            "notes": "Reference downloads and standard bioinformatics queries were permitted.",
            "reporting_status": "reported",
            "value": "canonical bioinformatics databases including NCBI and Ensembl"
          },
          "external_tools": {
            "notes": "Exact inventory not reported.",
            "reporting_status": "reported",
            "value": "preinstalled tools plus packages installable through pip and conda"
          },
          "internet": {
            "notes": "Canonical bioinformatics resources were allowed; source-dataset reverse lookup was prohibited.",
            "reporting_status": "reported",
            "value": "allowlisted external access"
          }
        },
        "turns": {
          "notes": "Open-ended tool-using trajectory ending in a final answer.",
          "reporting_status": "reported",
          "value": "multi-turn agent episode"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "biomysterybench-evidence-solvable-results"
          ],
          "metric_id": "episode-accuracy",
          "model_id": "claude-haiku-4-5",
          "n": 76,
          "notes": "380 model episodes; official figure labels the point estimate and draws unlabeled bootstrap error bars.",
          "status": "verified",
          "value": 36.8
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "biomysterybench-evidence-solvable-results"
          ],
          "metric_id": "episode-accuracy",
          "model_id": "claude-sonnet-4-6",
          "n": 76,
          "notes": "380 model episodes; official figure labels the point estimate and draws unlabeled bootstrap error bars.",
          "status": "verified",
          "value": 71.8
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "biomysterybench-evidence-solvable-results"
          ],
          "metric_id": "episode-accuracy",
          "model_id": "claude-opus-4-6",
          "n": 76,
          "notes": "380 model episodes; official figure labels the point estimate and draws unlabeled bootstrap error bars.",
          "status": "verified",
          "value": 77.4
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "biomysterybench-evidence-solvable-results"
          ],
          "metric_id": "episode-accuracy",
          "model_id": "claude-opus-4-7",
          "n": 76,
          "notes": "380 model episodes; official figure labels the point estimate and draws unlabeled bootstrap error bars.",
          "status": "verified",
          "value": 78.9
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "biomysterybench-evidence-solvable-results"
          ],
          "metric_id": "episode-accuracy",
          "model_id": "claude-mythos-preview",
          "n": 76,
          "notes": "380 model episodes; official figure labels the point estimate and draws unlabeled bootstrap error bars.",
          "status": "verified",
          "value": 82.6
        }
      ],
      "scope": {
        "filter": "Problems solved by at least one of up to five domain-expert human benchmarkers.",
        "n": 76,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "human-solvable",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "All five printed Figure 1 labels transcribed from the immutable official image; error-bar bounds are intentionally left null.",
        "status": "verified"
      },
      "work_id": "biomysterybench-official",
      "work_version_id": "biomysterybench-official-2026-04-29"
    },
    {
      "benchmark_id": "biosecbench-surveillance",
      "benchmark_version": "initial-release",
      "comparability_group": "biosecbench-8d53fd8-claude-code",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-08-11",
          "id": "biosecbench-8d53fd8-claude-code-protocol-evidence",
          "locator": {
            "note": "Full 100-evaluation scope, Claude Code harness, tools, three repeats, deterministic grading, aggregation, and confidence intervals.",
            "type": "repository-path",
            "value": "METHODS.md at commit 8d53fd8517cc74202eb18b618e8b39b4ffaf0c87"
          },
          "source_id": "biosecbench-surveillance-official-repository-resource",
          "source_type": "resource",
          "supports": [
            "/benchmark_version",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-08-11",
          "id": "biosecbench-8d53fd8-claude-code-results-evidence",
          "locator": {
            "note": "Exact model labels, endpoint pass rates, 95% confidence intervals, and gradable evaluation counts.",
            "type": "repository-path",
            "value": "results/config_results.csv; harness=claude-code at commit 8d53fd8517cc74202eb18b618e8b39b4ffaf0c87"
          },
          "source_id": "biosecbench-surveillance-official-repository-resource",
          "source_type": "resource",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "biosecbench-8d53fd8-claude-code",
      "metrics": [
        {
          "aggregation": "mean of per-evaluation pass rates; evaluations with no gradeable run are dropped",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "endpoint-pass-rate",
          "pass_threshold": null,
          "range": null,
          "source_label": "endpoint pass rate",
          "tolerance": null,
          "unit": "%"
        }
      ],
      "model_ids": [
        "anthropic-opus-4-6",
        "anthropic-opus-4-7",
        "anthropic-opus-4-8",
        "anthropic-sonnet-4-6"
      ],
      "protocol": {
        "contamination": {
          "notes": "No contamination or decontamination analysis is reported for these results.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": null,
          "model": null,
          "reporting_status": "reported",
          "type": "deterministic typed-field checks"
        },
        "reasoning": {
          "notes": "No model-specific reasoning-effort setting is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "repeats": {
          "notes": "Each model-by-harness configuration was run three times per evaluation.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Random seeds are not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The official repository does not report a demonstration count.",
          "reporting_status": "not_reported",
          "value": null
        },
        "statistical": {
          "notes": "METHODS.md defines aggregation and the confidence interval.",
          "reporting_status": "reported",
          "value": "95% Student-t confidence interval over evaluations"
        },
        "system_prompt_public": {
          "notes": "The exact system prompt and harness prompt configuration are not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "temperature": {
          "notes": "Temperature is not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "No per-evaluation time budget is reported in the pinned repository snapshot.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No token budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Browser availability is not separately reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "code_execution": {
            "notes": "Code or shell execution availability is not stated as a separate setting.",
            "reporting_status": "not_reported",
            "value": null
          },
          "container": {
            "notes": "METHODS.md states that all configurations used the same containerized sandbox.",
            "reporting_status": "reported",
            "value": "identical containerized sandbox"
          },
          "databases": {
            "notes": "METHODS.md reports access to resistance and virulence databases without exact database versions.",
            "reporting_status": "reported",
            "value": [
              "resistance databases",
              "virulence databases"
            ]
          },
          "external_tools": {
            "notes": "METHODS.md identifies the Claude Code harness and representative bioinformatics tools.",
            "reporting_status": "reported",
            "value": [
              "Claude Code",
              "assemblers",
              "aligners",
              "taxonomic classifiers",
              "variant callers"
            ]
          },
          "internet": {
            "notes": "METHODS.md states that agents had internet access.",
            "reporting_status": "reported",
            "value": true
          }
        },
        "turns": {
          "notes": "The exact turn or step limit is not reported in the pinned repository snapshot.",
          "reporting_status": "not_reported",
          "value": null
        }
      },
      "results": [
        {
          "ci_high": 56.9,
          "ci_low": 36.4,
          "confidence": "high",
          "evidence_ids": [
            "biosecbench-8d53fd8-claude-code-results-evidence"
          ],
          "metric_id": "endpoint-pass-rate",
          "model_id": "anthropic-opus-4-6",
          "n": 80,
          "notes": "Claude Code; n is the number of gradable evaluations after evaluations with no gradable run were dropped; the configured benchmark scope was all 100 evaluations.",
          "status": "verified",
          "value": 46.7
        },
        {
          "ci_high": 54.6,
          "ci_low": 33.7,
          "confidence": "high",
          "evidence_ids": [
            "biosecbench-8d53fd8-claude-code-results-evidence"
          ],
          "metric_id": "endpoint-pass-rate",
          "model_id": "anthropic-opus-4-7",
          "n": 77,
          "notes": "Claude Code; n is the number of gradable evaluations after evaluations with no gradable run were dropped; the configured benchmark scope was all 100 evaluations.",
          "status": "verified",
          "value": 44.2
        },
        {
          "ci_high": 53.2,
          "ci_low": 33.9,
          "confidence": "high",
          "evidence_ids": [
            "biosecbench-8d53fd8-claude-code-results-evidence"
          ],
          "metric_id": "endpoint-pass-rate",
          "model_id": "anthropic-sonnet-4-6",
          "n": 83,
          "notes": "Claude Code; n is the number of gradable evaluations after evaluations with no gradable run were dropped; the configured benchmark scope was all 100 evaluations.",
          "status": "verified",
          "value": 43.6
        },
        {
          "ci_high": 49.4,
          "ci_low": 29.5,
          "confidence": "high",
          "evidence_ids": [
            "biosecbench-8d53fd8-claude-code-results-evidence"
          ],
          "metric_id": "endpoint-pass-rate",
          "model_id": "anthropic-opus-4-8",
          "n": 76,
          "notes": "Claude Code; n is the number of gradable evaluations after evaluations with no gradable run were dropped; the configured benchmark scope was all 100 evaluations.",
          "status": "verified",
          "value": 39.5
        }
      ],
      "scope": {
        "filter": null,
        "n": 100,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-08-11",
        "notes": "Full-scope protocol and exact machine-readable result rows passed independent high-confidence verification.",
        "status": "verified"
      },
      "work_id": "biosecbench-surveillance-repository-result-snapshot",
      "work_version_id": "biosecbench-surveillance-repository-result-snapshot-8d53fd8"
    },
    {
      "benchmark_id": "biosecbench-surveillance",
      "benchmark_version": "initial-release",
      "comparability_group": "biosecbench-8d53fd8-openai-codex",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-08-11",
          "id": "biosecbench-8d53fd8-openai-codex-protocol-evidence",
          "locator": {
            "note": "Full 100-evaluation scope, OpenAI Codex harness, tools, three repeats, deterministic grading, aggregation, and confidence intervals.",
            "type": "repository-path",
            "value": "METHODS.md at commit 8d53fd8517cc74202eb18b618e8b39b4ffaf0c87"
          },
          "source_id": "biosecbench-surveillance-official-repository-resource",
          "source_type": "resource",
          "supports": [
            "/benchmark_version",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-08-11",
          "id": "biosecbench-8d53fd8-openai-codex-results-evidence",
          "locator": {
            "note": "Exact model labels, endpoint pass rates, 95% confidence intervals, and gradable evaluation counts.",
            "type": "repository-path",
            "value": "results/config_results.csv; harness=openai-codex at commit 8d53fd8517cc74202eb18b618e8b39b4ffaf0c87"
          },
          "source_id": "biosecbench-surveillance-official-repository-resource",
          "source_type": "resource",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "biosecbench-8d53fd8-openai-codex",
      "metrics": [
        {
          "aggregation": "mean of per-evaluation pass rates; evaluations with no gradeable run are dropped",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "endpoint-pass-rate",
          "pass_threshold": null,
          "range": null,
          "source_label": "endpoint pass rate",
          "tolerance": null,
          "unit": "%"
        }
      ],
      "model_ids": [
        "gpt-5-4",
        "gpt-5-5"
      ],
      "protocol": {
        "contamination": {
          "notes": "No contamination or decontamination analysis is reported for these results.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": null,
          "model": null,
          "reporting_status": "reported",
          "type": "deterministic typed-field checks"
        },
        "reasoning": {
          "notes": "No model-specific reasoning-effort setting is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "repeats": {
          "notes": "Each model-by-harness configuration was run three times per evaluation.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Random seeds are not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The official repository does not report a demonstration count.",
          "reporting_status": "not_reported",
          "value": null
        },
        "statistical": {
          "notes": "METHODS.md defines aggregation and the confidence interval.",
          "reporting_status": "reported",
          "value": "95% Student-t confidence interval over evaluations"
        },
        "system_prompt_public": {
          "notes": "The exact system prompt and harness prompt configuration are not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "temperature": {
          "notes": "Temperature is not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "No per-evaluation time budget is reported in the pinned repository snapshot.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No token budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Browser availability is not separately reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "code_execution": {
            "notes": "Code or shell execution availability is not stated as a separate setting.",
            "reporting_status": "not_reported",
            "value": null
          },
          "container": {
            "notes": "METHODS.md states that all configurations used the same containerized sandbox.",
            "reporting_status": "reported",
            "value": "identical containerized sandbox"
          },
          "databases": {
            "notes": "METHODS.md reports access to resistance and virulence databases without exact database versions.",
            "reporting_status": "reported",
            "value": [
              "resistance databases",
              "virulence databases"
            ]
          },
          "external_tools": {
            "notes": "METHODS.md identifies the OpenAI Codex harness and representative bioinformatics tools.",
            "reporting_status": "reported",
            "value": [
              "OpenAI Codex",
              "assemblers",
              "aligners",
              "taxonomic classifiers",
              "variant callers"
            ]
          },
          "internet": {
            "notes": "METHODS.md states that agents had internet access.",
            "reporting_status": "reported",
            "value": true
          }
        },
        "turns": {
          "notes": "The exact turn or step limit is not reported in the pinned repository snapshot.",
          "reporting_status": "not_reported",
          "value": null
        }
      },
      "results": [
        {
          "ci_high": 59.5,
          "ci_low": 40.8,
          "confidence": "high",
          "evidence_ids": [
            "biosecbench-8d53fd8-openai-codex-results-evidence"
          ],
          "metric_id": "endpoint-pass-rate",
          "model_id": "gpt-5-5",
          "n": 93,
          "notes": "OpenAI Codex; n is the number of gradable evaluations after evaluations with no gradable run were dropped; the configured benchmark scope was all 100 evaluations.",
          "status": "verified",
          "value": 50.2
        },
        {
          "ci_high": 50.1,
          "ci_low": 32.5,
          "confidence": "high",
          "evidence_ids": [
            "biosecbench-8d53fd8-openai-codex-results-evidence"
          ],
          "metric_id": "endpoint-pass-rate",
          "model_id": "gpt-5-4",
          "n": 92,
          "notes": "OpenAI Codex; n is the number of gradable evaluations after evaluations with no gradable run were dropped; the configured benchmark scope was all 100 evaluations.",
          "status": "verified",
          "value": 41.3
        }
      ],
      "scope": {
        "filter": null,
        "n": 100,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-08-11",
        "notes": "Full-scope protocol and exact machine-readable result rows passed independent high-confidence verification.",
        "status": "verified"
      },
      "work_id": "biosecbench-surveillance-repository-result-snapshot",
      "work_version_id": "biosecbench-surveillance-repository-result-snapshot-8d53fd8"
    },
    {
      "benchmark_id": "biosecbench-surveillance",
      "benchmark_version": "initial-release",
      "comparability_group": "biosecbench-8d53fd8-pi",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-08-11",
          "id": "biosecbench-8d53fd8-pi-protocol-evidence",
          "locator": {
            "note": "Full 100-evaluation scope, PI harness, tools, three repeats, deterministic grading, aggregation, and confidence intervals.",
            "type": "repository-path",
            "value": "METHODS.md at commit 8d53fd8517cc74202eb18b618e8b39b4ffaf0c87"
          },
          "source_id": "biosecbench-surveillance-official-repository-resource",
          "source_type": "resource",
          "supports": [
            "/benchmark_version",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-08-11",
          "id": "biosecbench-8d53fd8-pi-results-evidence",
          "locator": {
            "note": "Exact model labels, endpoint pass rates, 95% confidence intervals, and gradable evaluation counts.",
            "type": "repository-path",
            "value": "results/config_results.csv; harness=pi at commit 8d53fd8517cc74202eb18b618e8b39b4ffaf0c87"
          },
          "source_id": "biosecbench-surveillance-official-repository-resource",
          "source_type": "resource",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "biosecbench-8d53fd8-pi",
      "metrics": [
        {
          "aggregation": "mean of per-evaluation pass rates; evaluations with no gradeable run are dropped",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "endpoint-pass-rate",
          "pass_threshold": null,
          "range": null,
          "source_label": "endpoint pass rate",
          "tolerance": null,
          "unit": "%"
        }
      ],
      "model_ids": [
        "anthropic-opus-4-6",
        "anthropic-opus-4-7",
        "anthropic-opus-4-8",
        "anthropic-sonnet-4-6",
        "gemini-3-1-pro-preview",
        "gemini-3-5-flash",
        "gpt-5-4",
        "gpt-5-5",
        "grok-4-20-beta-0309-reasoning",
        "grok-4-3"
      ],
      "protocol": {
        "contamination": {
          "notes": "No contamination or decontamination analysis is reported for these results.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": null,
          "model": null,
          "reporting_status": "reported",
          "type": "deterministic typed-field checks"
        },
        "reasoning": {
          "notes": "No model-specific reasoning-effort setting is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "repeats": {
          "notes": "Each model-by-harness configuration was run three times per evaluation.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Random seeds are not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The official repository does not report a demonstration count.",
          "reporting_status": "not_reported",
          "value": null
        },
        "statistical": {
          "notes": "METHODS.md defines aggregation and the confidence interval.",
          "reporting_status": "reported",
          "value": "95% Student-t confidence interval over evaluations"
        },
        "system_prompt_public": {
          "notes": "The exact system prompt and harness prompt configuration are not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "temperature": {
          "notes": "Temperature is not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "No per-evaluation time budget is reported in the pinned repository snapshot.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No token budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Browser availability is not separately reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "code_execution": {
            "notes": "Code or shell execution availability is not stated as a separate setting.",
            "reporting_status": "not_reported",
            "value": null
          },
          "container": {
            "notes": "METHODS.md states that all configurations used the same containerized sandbox.",
            "reporting_status": "reported",
            "value": "identical containerized sandbox"
          },
          "databases": {
            "notes": "METHODS.md reports access to resistance and virulence databases without exact database versions.",
            "reporting_status": "reported",
            "value": [
              "resistance databases",
              "virulence databases"
            ]
          },
          "external_tools": {
            "notes": "METHODS.md identifies the PI harness and representative bioinformatics tools.",
            "reporting_status": "reported",
            "value": [
              "PI",
              "assemblers",
              "aligners",
              "taxonomic classifiers",
              "variant callers"
            ]
          },
          "internet": {
            "notes": "METHODS.md states that agents had internet access.",
            "reporting_status": "reported",
            "value": true
          }
        },
        "turns": {
          "notes": "The exact turn or step limit is not reported in the pinned repository snapshot.",
          "reporting_status": "not_reported",
          "value": null
        }
      },
      "results": [
        {
          "ci_high": 60.2,
          "ci_low": 40.2,
          "confidence": "high",
          "evidence_ids": [
            "biosecbench-8d53fd8-pi-results-evidence"
          ],
          "metric_id": "endpoint-pass-rate",
          "model_id": "anthropic-opus-4-8",
          "n": 83,
          "notes": "PI; n is the number of gradable evaluations after evaluations with no gradable run were dropped; the configured benchmark scope was all 100 evaluations.",
          "status": "verified",
          "value": 50.2
        },
        {
          "ci_high": 54.6,
          "ci_low": 35.0,
          "confidence": "high",
          "evidence_ids": [
            "biosecbench-8d53fd8-pi-results-evidence"
          ],
          "metric_id": "endpoint-pass-rate",
          "model_id": "gpt-5-5",
          "n": 77,
          "notes": "PI; n is the number of gradable evaluations after evaluations with no gradable run were dropped; the configured benchmark scope was all 100 evaluations.",
          "status": "verified",
          "value": 44.8
        },
        {
          "ci_high": 59.1,
          "ci_low": 40.0,
          "confidence": "high",
          "evidence_ids": [
            "biosecbench-8d53fd8-pi-results-evidence"
          ],
          "metric_id": "endpoint-pass-rate",
          "model_id": "anthropic-opus-4-7",
          "n": 78,
          "notes": "PI; n is the number of gradable evaluations after evaluations with no gradable run were dropped; the configured benchmark scope was all 100 evaluations.",
          "status": "verified",
          "value": 49.6
        },
        {
          "ci_high": 58.2,
          "ci_low": 39.0,
          "confidence": "high",
          "evidence_ids": [
            "biosecbench-8d53fd8-pi-results-evidence"
          ],
          "metric_id": "endpoint-pass-rate",
          "model_id": "anthropic-sonnet-4-6",
          "n": 83,
          "notes": "PI; n is the number of gradable evaluations after evaluations with no gradable run were dropped; the configured benchmark scope was all 100 evaluations.",
          "status": "verified",
          "value": 48.6
        },
        {
          "ci_high": 56.7,
          "ci_low": 37.5,
          "confidence": "high",
          "evidence_ids": [
            "biosecbench-8d53fd8-pi-results-evidence"
          ],
          "metric_id": "endpoint-pass-rate",
          "model_id": "gemini-3-5-flash",
          "n": 92,
          "notes": "PI; n is the number of gradable evaluations after evaluations with no gradable run were dropped; the configured benchmark scope was all 100 evaluations.",
          "status": "verified",
          "value": 47.1
        },
        {
          "ci_high": 55.4,
          "ci_low": 35.9,
          "confidence": "high",
          "evidence_ids": [
            "biosecbench-8d53fd8-pi-results-evidence"
          ],
          "metric_id": "endpoint-pass-rate",
          "model_id": "anthropic-opus-4-6",
          "n": 81,
          "notes": "PI; n is the number of gradable evaluations after evaluations with no gradable run were dropped; the configured benchmark scope was all 100 evaluations.",
          "status": "verified",
          "value": 45.7
        },
        {
          "ci_high": 53.8,
          "ci_low": 36.7,
          "confidence": "high",
          "evidence_ids": [
            "biosecbench-8d53fd8-pi-results-evidence"
          ],
          "metric_id": "endpoint-pass-rate",
          "model_id": "gemini-3-1-pro-preview",
          "n": 95,
          "notes": "PI; n is the number of gradable evaluations after evaluations with no gradable run were dropped; the configured benchmark scope was all 100 evaluations.",
          "status": "verified",
          "value": 45.3
        },
        {
          "ci_high": 47.6,
          "ci_low": 29.3,
          "confidence": "high",
          "evidence_ids": [
            "biosecbench-8d53fd8-pi-results-evidence"
          ],
          "metric_id": "endpoint-pass-rate",
          "model_id": "gpt-5-4",
          "n": 81,
          "notes": "PI; n is the number of gradable evaluations after evaluations with no gradable run were dropped; the configured benchmark scope was all 100 evaluations.",
          "status": "verified",
          "value": 38.5
        },
        {
          "ci_high": 21.9,
          "ci_low": 10.1,
          "confidence": "high",
          "evidence_ids": [
            "biosecbench-8d53fd8-pi-results-evidence"
          ],
          "metric_id": "endpoint-pass-rate",
          "model_id": "grok-4-3",
          "n": 100,
          "notes": "PI; n is the number of gradable evaluations after evaluations with no gradable run were dropped; the configured benchmark scope was all 100 evaluations.",
          "status": "verified",
          "value": 16.0
        },
        {
          "ci_high": 19.2,
          "ci_low": 8.1,
          "confidence": "high",
          "evidence_ids": [
            "biosecbench-8d53fd8-pi-results-evidence"
          ],
          "metric_id": "endpoint-pass-rate",
          "model_id": "grok-4-20-beta-0309-reasoning",
          "n": 100,
          "notes": "PI; n is the number of gradable evaluations after evaluations with no gradable run were dropped; the configured benchmark scope was all 100 evaluations.",
          "status": "verified",
          "value": 13.7
        }
      ],
      "scope": {
        "filter": null,
        "n": 100,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-08-11",
        "notes": "Full-scope protocol and exact machine-readable result rows passed independent high-confidence verification.",
        "status": "verified"
      },
      "work_id": "biosecbench-surveillance-repository-result-snapshot",
      "work_version_id": "biosecbench-surveillance-repository-result-snapshot-8d53fd8"
    },
    {
      "benchmark_id": "bixbench",
      "benchmark_version": "v1.0",
      "comparability_group": "bixbench-v1-open-agentic-images-ten-runs",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bixbench-evidence-paper-open-scope-protocol",
          "locator": {
            "note": "Reports 296 questions, Docker environment, three tools, notebook execution, ten parallel analyses, plots/images condition, Claude judge, and aggregation.",
            "type": "page",
            "value": "PDF pp. 4–7, §§3.2.2–3.2.5 and §4.1"
          },
          "source_id": "bixbench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bixbench-evidence-paper-open-config",
          "locator": {
            "note": "Records ReActAgent, temperature 1.0, maximum 25 steps, public prompt key, and total_questions 296.",
            "type": "repository-path",
            "value": "bixbench/config.yaml at commit 6c28217959d5d7dd6f48c59894534fced7c6c040"
          },
          "source_id": "bixbench-v1-paper-code-resource",
          "source_type": "resource",
          "supports": [
            "/protocol/reasoning",
            "/protocol/system_prompt_public",
            "/protocol/time_budget",
            "/protocol/temperature"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bixbench-evidence-paper-open-results",
          "locator": {
            "note": "Text explicitly reports Claude 3.5 Sonnet at 17% and GPT-4o at 9% in open-answer evaluation.",
            "type": "page",
            "value": "PDF pp. 1 and 6, abstract and §4.1; Figure 4"
          },
          "source_id": "bixbench-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bixbench-creator-paper",
      "metrics": [
        {
          "aggregation": "question-trajectory-weighted mean over ten runs per question",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Accuracy",
          "tolerance": "binary LLM judge",
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bixbench-claude-3-5-sonnet-unversioned",
        "bixbench-gpt-4o-unversioned"
      ],
      "protocol": {
        "contamination": {
          "notes": "No pretraining-contamination or decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": "Claude 3.5 Sonnet (exact version not reported)",
          "reporting_status": "reported",
          "type": "binary LLM judgment against the ground-truth solution"
        },
        "reasoning": {
          "notes": "Maximum steps are recorded in the paper-era config; no provider reasoning-effort control is reported.",
          "reporting_status": "reported",
          "value": "Aviary ReActAgent with at most 25 steps"
        },
        "repeats": {
          "notes": "Each capsule analysis was run independently ten times per model and modality.",
          "reporting_status": "reported",
          "value": 10
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Each trajectory starts from a task capsule without in-context demonstrations.",
          "reporting_status": "reported",
          "value": "zero-shot task initialization"
        },
        "statistical": {
          "notes": "The paper text prints rounded headline accuracies but not numeric interval bounds.",
          "reporting_status": "reported",
          "value": "accuracy over all question-trajectory pairs; official plots use 95% Wilson intervals"
        },
        "system_prompt_public": {
          "notes": "The final trajectory-initiation prompt is reproduced in Appendix A and the paper-era harness is public.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Paper-era official configuration value.",
          "reporting_status": "reported",
          "value": 1.0
        },
        "time_budget": {
          "notes": "No wall-clock or token limit is reported.",
          "reporting_status": "reported",
          "value": "maximum 25 agent steps"
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "The published scaffold exposes notebook editing, workspace listing, and answer submission rather than a browser.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Every notebook edit reruns the notebook with Python, R, and Bash packages available.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "Analyses run in the pre-built BixBench-env:v1.0 Docker container.",
            "reporting_status": "reported",
            "value": true
          },
          "databases": {
            "notes": "External database access is not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "external_tools": {
            "notes": "These are the three agent tools listed in the paper.",
            "reporting_status": "reported",
            "value": "edit cell, list workdir, and submit answer"
          },
          "internet": {
            "notes": "Network availability inside the container is not reported.",
            "reporting_status": "not_reported",
            "value": null
          }
        },
        "turns": {
          "notes": "The agent repeatedly edits and executes a notebook, inspects the workspace, and submits a final answer.",
          "reporting_status": "reported",
          "value": "multi-turn ReAct agent trajectory"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bixbench-evidence-paper-open-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bixbench-gpt-4o-unversioned",
          "n": 296,
          "notes": "Rounded open-answer headline in the paper text; ten trajectories per capsule, with plots/images allowed.",
          "status": "verified",
          "value": 9.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bixbench-evidence-paper-open-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bixbench-claude-3-5-sonnet-unversioned",
          "n": 296,
          "notes": "Rounded open-answer headline in the paper text; ten trajectories per capsule, with plots/images allowed.",
          "status": "verified",
          "value": 17.0
        }
      ],
      "scope": {
        "filter": "All 296 v1.0 questions in the primary open-answer, plots/images-allowed condition.",
        "n": 296,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Original full scope, tool/container protocol, ten repeats, grader, and the two printed open-answer scores verified; unprinted interval bounds are left null.",
        "status": "verified"
      },
      "work_id": "bixbench-paper",
      "work_version_id": "bixbench-paper-2025-02-28"
    },
    {
      "benchmark_id": "bixbench",
      "benchmark_version": "v1.0",
      "comparability_group": "bixbench-v1-mcq-refusal-no-images-ten-runs",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bixbench-evidence-paper-no-images",
          "locator": {
            "note": "States that the refusal option is present and agents are instructed not to produce images/plots.",
            "type": "figure",
            "value": "PDF pp. 7–8, Figure 5 caption and image-generation ablation discussion"
          },
          "source_id": "bixbench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        }
      ],
      "id": "bixbench-paper-mcq-no-images",
      "metrics": [
        {
          "aggregation": "majority vote over ten trajectories per question",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "majority-vote-accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Accuracy",
          "tolerance": "exact selected option",
          "unit": "percent"
        },
        {
          "aggregation": "correct among questions not assigned refusal",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Precision",
          "tolerance": "exact selected option",
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bixbench-claude-3-5-sonnet-unversioned",
        "bixbench-gpt-4o-unversioned"
      ],
      "protocol": {
        "contamination": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": "Claude 3.5 Sonnet (exact version not reported)",
          "reporting_status": "reported",
          "type": "second-LLM multiple-choice selection with Insufficient information refusal"
        },
        "reasoning": {
          "notes": "No provider effort control reported.",
          "reporting_status": "reported",
          "value": "Aviary ReActAgent with at most 25 steps and an instruction to avoid plots/images"
        },
        "repeats": {
          "notes": "Ten trajectories feed majority voting.",
          "reporting_status": "reported",
          "value": 10
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "No demonstrations.",
          "reporting_status": "reported",
          "value": "zero-shot task initialization"
        },
        "statistical": {
          "notes": "Figure 5 reports trajectories but no printed aggregate scalar, so no value is transcribed.",
          "reporting_status": "reported",
          "value": "majority-vote accuracy and precision over ten trajectories"
        },
        "system_prompt_public": {
          "notes": "Prompts and paper-era harness are public.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Paper-era official configuration value.",
          "reporting_status": "reported",
          "value": 1.0
        },
        "time_budget": {
          "notes": "No wall-clock limit reported.",
          "reporting_status": "reported",
          "value": "maximum 25 agent steps"
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "No browser tool.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Notebook code execution remains enabled.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "BixBench-env:v1.0 Docker container.",
            "reporting_status": "reported",
            "value": true
          },
          "databases": {
            "notes": "Not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "external_tools": {
            "notes": "Official scaffold tools.",
            "reporting_status": "reported",
            "value": "edit cell, list workdir, and submit answer"
          },
          "internet": {
            "notes": "Not reported.",
            "reporting_status": "not_reported",
            "value": null
          }
        },
        "turns": {
          "notes": "Same agentic analysis except for the no-plots instruction.",
          "reporting_status": "reported",
          "value": "multi-turn ReAct analysis followed by a separate MCQ call"
        }
      },
      "results": [],
      "scope": {
        "filter": "All v1.0 questions in the MCQ-with-refusal ablation where agents were instructed not to generate plots/images.",
        "n": 296,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "No-image protocol verified; plotted trajectories are not digitized into registry values.",
        "status": "verified"
      },
      "work_id": "bixbench-paper",
      "work_version_id": "bixbench-paper-2025-02-28"
    },
    {
      "benchmark_id": "bixbench",
      "benchmark_version": "v1.0",
      "comparability_group": "bixbench-v1-mcq-no-refusal-images-ten-runs",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bixbench-evidence-paper-mcq-no-refusal",
          "locator": {
            "note": "Reports the forced-answer ablation and ten-trajectory majority-vote accuracy.",
            "type": "page",
            "value": "PDF pp. 6–7, Figure 4 and §4.1"
          },
          "source_id": "bixbench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        }
      ],
      "id": "bixbench-paper-mcq-no-refusal",
      "metrics": [
        {
          "aggregation": "majority vote over ten trajectories per question",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "majority-vote-accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Accuracy",
          "tolerance": "exact selected option",
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bixbench-claude-3-5-sonnet-unversioned",
        "bixbench-gpt-4o-unversioned"
      ],
      "protocol": {
        "contamination": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": "Claude 3.5 Sonnet (exact version not reported)",
          "reporting_status": "reported",
          "type": "second-LLM forced multiple-choice selection"
        },
        "reasoning": {
          "notes": "No provider effort control reported.",
          "reporting_status": "reported",
          "value": "Aviary ReActAgent with at most 25 steps"
        },
        "repeats": {
          "notes": "Ten trajectories feed majority voting.",
          "reporting_status": "reported",
          "value": 10
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "No demonstrations.",
          "reporting_status": "reported",
          "value": "zero-shot task initialization"
        },
        "statistical": {
          "notes": "Figure 4 has bars but no printed scalar labels, so no values are transcribed.",
          "reporting_status": "reported",
          "value": "majority vote over ten trajectories with accuracy"
        },
        "system_prompt_public": {
          "notes": "Prompts are public.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Paper-era official configuration value.",
          "reporting_status": "reported",
          "value": 1.0
        },
        "time_budget": {
          "notes": "No wall-clock limit reported.",
          "reporting_status": "reported",
          "value": "maximum 25 agent steps"
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "No browser tool.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Notebook code execution.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "BixBench-env:v1.0 Docker container.",
            "reporting_status": "reported",
            "value": true
          },
          "databases": {
            "notes": "Not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "external_tools": {
            "notes": "Official scaffold tools.",
            "reporting_status": "reported",
            "value": "edit cell, list workdir, and submit answer"
          },
          "internet": {
            "notes": "Not reported.",
            "reporting_status": "not_reported",
            "value": null
          }
        },
        "turns": {
          "notes": "The second LLM receives notebook context and must choose a substantive option.",
          "reporting_status": "reported",
          "value": "multi-turn ReAct analysis followed by a separate MCQ call"
        }
      },
      "results": [],
      "scope": {
        "filter": "All v1.0 questions, plots/images allowed, without the refusal option.",
        "n": 296,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Protocol verified; numeric bars are intentionally not estimated from Figure 4.",
        "status": "verified"
      },
      "work_id": "bixbench-paper",
      "work_version_id": "bixbench-paper-2025-02-28"
    },
    {
      "benchmark_id": "bixbench",
      "benchmark_version": "v1.0",
      "comparability_group": "bixbench-v1-mcq-refusal-images-ten-runs",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bixbench-evidence-paper-mcq-refusal",
          "locator": {
            "note": "Defines the second-LLM MCQ conversion, refusal option, ten-run majority vote, accuracy, and precision.",
            "type": "page",
            "value": "PDF pp. 5–7, §3.2.4, Figure 4, and §4.1"
          },
          "source_id": "bixbench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        }
      ],
      "id": "bixbench-paper-mcq-refusal",
      "metrics": [
        {
          "aggregation": "majority vote over ten trajectories per question",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "majority-vote-accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Accuracy",
          "tolerance": "exact selected option",
          "unit": "percent"
        },
        {
          "aggregation": "correct among questions not assigned the refusal option",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Precision",
          "tolerance": "exact selected option",
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bixbench-claude-3-5-sonnet-unversioned",
        "bixbench-gpt-4o-unversioned"
      ],
      "protocol": {
        "contamination": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": "Claude 3.5 Sonnet (exact version not reported)",
          "reporting_status": "reported",
          "type": "second-LLM multiple-choice selection with Insufficient information refusal"
        },
        "reasoning": {
          "notes": "No provider effort control is reported.",
          "reporting_status": "reported",
          "value": "Aviary ReActAgent with at most 25 steps"
        },
        "repeats": {
          "notes": "Ten trajectories feed majority voting.",
          "reporting_status": "reported",
          "value": 10
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "No demonstrations.",
          "reporting_status": "reported",
          "value": "zero-shot task initialization"
        },
        "statistical": {
          "notes": "Figure 4 has bars but no printed scalar labels, so no values are transcribed.",
          "reporting_status": "reported",
          "value": "majority vote over ten trajectories; accuracy and precision among non-refusal answers"
        },
        "system_prompt_public": {
          "notes": "Prompts are in the paper appendix and official harness.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Paper-era official configuration value.",
          "reporting_status": "reported",
          "value": 1.0
        },
        "time_budget": {
          "notes": "No wall-clock limit reported.",
          "reporting_status": "reported",
          "value": "maximum 25 agent steps"
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "No browser tool.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Notebook code execution in Python, R, and Bash.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "BixBench-env:v1.0 Docker container.",
            "reporting_status": "reported",
            "value": true
          },
          "databases": {
            "notes": "Not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "external_tools": {
            "notes": "Official scaffold tools.",
            "reporting_status": "reported",
            "value": "edit cell, list workdir, and submit answer"
          },
          "internet": {
            "notes": "Not reported.",
            "reporting_status": "not_reported",
            "value": null
          }
        },
        "turns": {
          "notes": "The second LLM receives the completed notebook, original response, question, and options.",
          "reporting_status": "reported",
          "value": "multi-turn ReAct analysis followed by a separate MCQ call"
        }
      },
      "results": [],
      "scope": {
        "filter": "All v1.0 questions, plots/images allowed, with an Insufficient information option.",
        "n": 296,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Protocol verified; numeric bars are intentionally not estimated from Figure 4.",
        "status": "verified"
      },
      "work_id": "bixbench-paper",
      "work_version_id": "bixbench-paper-2025-02-28"
    },
    {
      "benchmark_id": "bixbench",
      "benchmark_version": "v1.5",
      "comparability_group": "bixbench-v1-5-agentic-mcq-no-refusal-images-five-replicas",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bixbench-evidence-v1-5-agentic-mcq-no-refusal-images",
          "locator": {
            "note": "Defines images allowed, forced-choice condition, full dataset, agent settings, metric, and majority-vote postprocessing.",
            "type": "repository-path",
            "value": "image model configs, v1.5_paper_results.yaml, README.md, and scripts/run_agentic.sh at commit 49311180bdacb324c596f2e07596c126f2004008"
          },
          "source_id": "bixbench-repository-resource",
          "source_type": "resource",
          "supports": [
            "/benchmark_version",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        }
      ],
      "id": "bixbench-v1-5-agentic-mcq-no-refusal-images",
      "metrics": [
        {
          "aggregation": "question-replica-weighted mean",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Accuracy",
          "tolerance": "exact selected option",
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bixbench-claude-3-5-sonnet-20241022",
        "bixbench-gpt-4o-unversioned"
      ],
      "protocol": {
        "contamination": {
          "notes": "No model-specific decontamination analysis.",
          "reporting_status": "reported",
          "value": "public benchmark canary"
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "forced multiple-choice conversion without refusal"
        },
        "reasoning": {
          "notes": "No provider reasoning-effort control.",
          "reporting_status": "reported",
          "value": "SimpleAgent with at most 20 steps"
        },
        "repeats": {
          "notes": "README and postprocessing state five replicas; the reproduction shell loop would request IDs 0–5 if run literally.",
          "reporting_status": "reported",
          "value": 5
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "No demonstrations.",
          "reporting_status": "reported",
          "value": "zero-shot task initialization"
        },
        "statistical": {
          "notes": "Official plots have no printed scalar labels.",
          "reporting_status": "reported",
          "value": "accuracy with 95% Wilson interval and majority-vote analysis"
        },
        "system_prompt_public": {
          "notes": "Public system and MCQ prompt templates.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Both official model configs.",
          "reporting_status": "reported",
          "value": 1.0
        },
        "time_budget": {
          "notes": "No per-episode wall time reported.",
          "reporting_status": "reported",
          "value": "maximum 20 agent steps"
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "No browser tool.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Mutable notebook execution.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "Official Docker notebook environment.",
            "reporting_status": "reported",
            "value": true
          },
          "databases": {
            "notes": "External database access is not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "external_tools": {
            "notes": "Plots and images are allowed.",
            "reporting_status": "reported",
            "value": "notebook editing and capsule workspace inspection"
          },
          "internet": {
            "notes": "Container network availability is not reported.",
            "reporting_status": "not_reported",
            "value": null
          }
        },
        "turns": {
          "notes": "Up to 20 agent steps before choice conversion.",
          "reporting_status": "reported",
          "value": "multi-turn Aviary SimpleAgent trajectory followed by forced-choice MCQ postprocessing"
        }
      },
      "results": [],
      "scope": {
        "filter": "All v1.5 questions, plots/images allowed, without the refusal option.",
        "n": 205,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Protocol and plotted metric verified; no bar height is inferred as a scalar.",
        "status": "verified"
      },
      "work_id": "bixbench-v1-5-release",
      "work_version_id": "bixbench-v1-5-release-2025-09-26"
    },
    {
      "benchmark_id": "bixbench",
      "benchmark_version": "v1.5",
      "comparability_group": "bixbench-v1-5-agentic-mcq-refusal-images-five-replicas",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bixbench-evidence-v1-5-agentic-mcq-refusal-images",
          "locator": {
            "note": "Defines images allowed, refusal condition, full dataset, agent settings, metric, and majority-vote postprocessing.",
            "type": "repository-path",
            "value": "image model configs, v1.5_paper_results.yaml, README.md, and scripts/run_agentic.sh at commit 49311180bdacb324c596f2e07596c126f2004008"
          },
          "source_id": "bixbench-repository-resource",
          "source_type": "resource",
          "supports": [
            "/benchmark_version",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        }
      ],
      "id": "bixbench-v1-5-agentic-mcq-refusal-images",
      "metrics": [
        {
          "aggregation": "question-replica-weighted mean",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Accuracy",
          "tolerance": "exact selected option",
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bixbench-claude-3-5-sonnet-20241022",
        "bixbench-gpt-4o-unversioned"
      ],
      "protocol": {
        "contamination": {
          "notes": "No model-specific decontamination analysis.",
          "reporting_status": "reported",
          "value": "public benchmark canary"
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "multiple-choice conversion with Insufficient information refusal"
        },
        "reasoning": {
          "notes": "No provider reasoning-effort control.",
          "reporting_status": "reported",
          "value": "SimpleAgent with at most 20 steps"
        },
        "repeats": {
          "notes": "README and postprocessing state five replicas; the reproduction shell loop would request IDs 0–5 if run literally.",
          "reporting_status": "reported",
          "value": 5
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "No demonstrations.",
          "reporting_status": "reported",
          "value": "zero-shot task initialization"
        },
        "statistical": {
          "notes": "Official plots have no printed scalar labels.",
          "reporting_status": "reported",
          "value": "accuracy with 95% Wilson interval and majority-vote analysis"
        },
        "system_prompt_public": {
          "notes": "Public system and MCQ prompt templates.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Both official model configs.",
          "reporting_status": "reported",
          "value": 1.0
        },
        "time_budget": {
          "notes": "No per-episode wall time reported.",
          "reporting_status": "reported",
          "value": "maximum 20 agent steps"
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "No browser tool.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Mutable notebook execution.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "Official Docker notebook environment.",
            "reporting_status": "reported",
            "value": true
          },
          "databases": {
            "notes": "External database access is not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "external_tools": {
            "notes": "Plots and images are allowed.",
            "reporting_status": "reported",
            "value": "notebook editing and capsule workspace inspection"
          },
          "internet": {
            "notes": "Container network availability is not reported.",
            "reporting_status": "not_reported",
            "value": null
          }
        },
        "turns": {
          "notes": "Up to 20 agent steps before choice conversion.",
          "reporting_status": "reported",
          "value": "multi-turn Aviary SimpleAgent trajectory followed by MCQ postprocessing"
        }
      },
      "results": [],
      "scope": {
        "filter": "All v1.5 questions, plots/images allowed, with the refusal option.",
        "n": 205,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Protocol and plotted metric verified; no bar height is inferred as a scalar.",
        "status": "verified"
      },
      "work_id": "bixbench-v1-5-release",
      "work_version_id": "bixbench-v1-5-release-2025-09-26"
    },
    {
      "benchmark_id": "bixbench",
      "benchmark_version": "v1.5",
      "comparability_group": "bixbench-v1-5-agentic-mcq-refusal-no-images-five-replicas",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bixbench-evidence-v1-5-agentic-mcq-refusal-no-images",
          "locator": {
            "note": "Defines avoid_images true, refusal condition, full dataset, agent settings, and majority-vote postprocessing.",
            "type": "repository-path",
            "value": "4o_no_image.yaml, claude_no_image.yaml, v1.5_paper_results.yaml, README.md, and scripts/run_agentic.sh at commit 49311180bdacb324c596f2e07596c126f2004008"
          },
          "source_id": "bixbench-repository-resource",
          "source_type": "resource",
          "supports": [
            "/benchmark_version",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        }
      ],
      "id": "bixbench-v1-5-agentic-mcq-refusal-no-images",
      "metrics": [
        {
          "aggregation": "majority vote over stated five replicas per question",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "majority-vote-accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Accuracy",
          "tolerance": "exact selected option",
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bixbench-claude-3-5-sonnet-20241022",
        "bixbench-gpt-4o-unversioned"
      ],
      "protocol": {
        "contamination": {
          "notes": "No model-specific decontamination analysis.",
          "reporting_status": "reported",
          "value": "public benchmark canary"
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "multiple-choice conversion with Insufficient information refusal"
        },
        "reasoning": {
          "notes": "No provider effort control.",
          "reporting_status": "reported",
          "value": "SimpleAgent with at most 20 steps and an instruction to avoid plots/images"
        },
        "repeats": {
          "notes": "README and postprocessing state five replicas; the reproduction shell loop would request IDs 0–5 if run literally.",
          "reporting_status": "reported",
          "value": 5
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "No demonstrations.",
          "reporting_status": "reported",
          "value": "zero-shot task initialization"
        },
        "statistical": {
          "notes": "Official trajectory plot has no printed aggregate scalar label.",
          "reporting_status": "reported",
          "value": "majority-vote accuracy by vote count"
        },
        "system_prompt_public": {
          "notes": "Public system and MCQ prompt templates.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Both official model configs.",
          "reporting_status": "reported",
          "value": 1.0
        },
        "time_budget": {
          "notes": "No per-episode wall time reported.",
          "reporting_status": "reported",
          "value": "maximum 20 agent steps"
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "No browser tool.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Mutable notebook execution remains enabled.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "Official Docker notebook environment.",
            "reporting_status": "reported",
            "value": true
          },
          "databases": {
            "notes": "External database access is not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "external_tools": {
            "notes": "Agent is explicitly instructed not to generate plots/images.",
            "reporting_status": "reported",
            "value": "notebook editing and capsule workspace inspection"
          },
          "internet": {
            "notes": "Container network availability is not reported.",
            "reporting_status": "not_reported",
            "value": null
          }
        },
        "turns": {
          "notes": "Up to 20 agent steps before choice conversion.",
          "reporting_status": "reported",
          "value": "multi-turn Aviary SimpleAgent trajectory followed by MCQ postprocessing"
        }
      },
      "results": [],
      "scope": {
        "filter": "All v1.5 questions, no plots/images, with the refusal option.",
        "n": 205,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "No-image protocol and trajectory plot verified; no curve value is digitized.",
        "status": "verified"
      },
      "work_id": "bixbench-v1-5-release",
      "work_version_id": "bixbench-v1-5-release-2025-09-26"
    },
    {
      "benchmark_id": "bixbench",
      "benchmark_version": "v1.5",
      "comparability_group": "bixbench-v1-5-agentic-open-images-five-replicas",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bixbench-evidence-v1-5-agentic-open-config",
          "locator": {
            "note": "Reports 205-question v1.5, Docker, SimpleAgent, 20 steps, temperature 1, images allowed, model strings, and stated five replicas.",
            "type": "repository-path",
            "value": "README.md; bixbench/run_configuration/4o_image.yaml and claude_image.yaml; scripts/run_agentic.sh at commit 49311180bdacb324c596f2e07596c126f2004008"
          },
          "source_id": "bixbench-repository-resource",
          "source_type": "resource",
          "supports": [
            "/benchmark_version",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bixbench-evidence-v1-5-agentic-open-figure",
          "locator": {
            "note": "Official accuracy bars with 95% Wilson intervals and no printed scalar labels.",
            "type": "figure",
            "value": "bixbench_results_comparison.png, Open-answer group"
          },
          "source_id": "bixbench-v1-5-results-resource",
          "source_type": "resource",
          "supports": [
            "/metrics"
          ]
        }
      ],
      "id": "bixbench-v1-5-agentic-open-images",
      "metrics": [
        {
          "aggregation": "question-replica-weighted mean",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Accuracy",
          "tolerance": "verifier-specific",
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bixbench-claude-3-5-sonnet-20241022",
        "bixbench-gpt-4o-unversioned"
      ],
      "protocol": {
        "contamination": {
          "notes": "No model-specific decontamination analysis.",
          "reporting_status": "reported",
          "value": "public benchmark canary"
        },
        "grader": {
          "human_review": false,
          "model": "GPT-4o for llm_verifier rows (exact endpoint not reported)",
          "reporting_status": "reported",
          "type": "row-specific open-answer verifier"
        },
        "reasoning": {
          "notes": "No provider reasoning-effort control.",
          "reporting_status": "reported",
          "value": "SimpleAgent with at most 20 steps"
        },
        "repeats": {
          "notes": "README and postprocessing configuration state five replicas; the reproduction shell loop is off by one if run literally and would request replica IDs 0–5.",
          "reporting_status": "reported",
          "value": 5
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "No demonstrations.",
          "reporting_status": "reported",
          "value": "zero-shot task initialization"
        },
        "statistical": {
          "notes": "The official PNG has no printed scalar labels; bars and interval endpoints are not digitized.",
          "reporting_status": "reported",
          "value": "accuracy with 95% Wilson interval"
        },
        "system_prompt_public": {
          "notes": "Public system prompt and open-answer prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Both official model configs.",
          "reporting_status": "reported",
          "value": 1.0
        },
        "time_budget": {
          "notes": "README estimates 24–48 hours for the full batch but no per-episode wall time.",
          "reporting_status": "reported",
          "value": "maximum 20 agent steps"
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "No browser tool in the published scaffold.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Mutable Python notebook with Python, R, and Bash packages.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "Official Docker notebook environment.",
            "reporting_status": "reported",
            "value": true
          },
          "databases": {
            "notes": "External database access is not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "external_tools": {
            "notes": "Plots and images are allowed in this condition.",
            "reporting_status": "reported",
            "value": "notebook editing and capsule workspace inspection"
          },
          "internet": {
            "notes": "Container network availability is not reported.",
            "reporting_status": "not_reported",
            "value": null
          }
        },
        "turns": {
          "notes": "Up to 20 steps over a mutable notebook and capsule workspace.",
          "reporting_status": "reported",
          "value": "multi-turn Aviary SimpleAgent trajectory"
        }
      },
      "results": [],
      "scope": {
        "filter": "All v1.5 questions in the plots/images-allowed open-answer condition.",
        "n": 205,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Scope, configs, and plotted metric verified; scalar scores are omitted because the figure does not print them. The upstream replica-loop inconsistency is retained in notes.",
        "status": "verified"
      },
      "work_id": "bixbench-v1-5-release",
      "work_version_id": "bixbench-v1-5-release-2025-09-26"
    },
    {
      "benchmark_id": "bixbench",
      "benchmark_version": "v1.5",
      "comparability_group": "bixbench-v1-5-zero-shot-mcq-no-refusal-temperature-one",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bixbench-evidence-v1-5-mcq-no-refusal-protocol",
          "locator": {
            "note": "Defines full-set zero-shot forced-choice calls, randomized options, temperature 1.0, and exact option grading.",
            "type": "repository-path",
            "value": "scripts/run_zeroshot.sh, generate_zeroshot_evals.py, bixbench/zero_shot.py, and grade_outputs.py at commit 49311180bdacb324c596f2e07596c126f2004008"
          },
          "source_id": "bixbench-repository-resource",
          "source_type": "resource",
          "supports": [
            "/benchmark_version",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bixbench-evidence-v1-5-mcq-no-refusal-results",
          "locator": {
            "note": "Exact accuracy, precision, coverage, n_total, n_correct, and n_sure for both forced-choice baselines.",
            "type": "repository-path",
            "value": "bixbench-v1.5_results/zero_shot_baselines.json at commit 28909d842bc492ecd99bab303279afb29e3cb353"
          },
          "source_id": "bixbench-v1-5-results-resource",
          "source_type": "resource",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bixbench-v1-5-zero-shot-mcq-no-refusal",
      "metrics": [
        {
          "aggregation": "correct over all questions",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "accuracy",
          "tolerance": "exact normalized option",
          "unit": "percent"
        },
        {
          "aggregation": "correct over sure answers",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "precision",
          "tolerance": "exact normalized option",
          "unit": "percent"
        },
        {
          "aggregation": "sure answers over all questions",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "coverage",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "coverage",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bixbench-claude-3-5-sonnet-unversioned",
        "bixbench-gpt-4o-unversioned"
      ],
      "protocol": {
        "contamination": {
          "notes": "No model-specific decontamination analysis.",
          "reporting_status": "reported",
          "value": "public benchmark canary"
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "exact normalized selected-option match"
        },
        "reasoning": {
          "notes": "No explicit reasoning-effort parameter.",
          "reporting_status": "reported",
          "value": "default model behavior"
        },
        "repeats": {
          "notes": "One prediction per model and question.",
          "reporting_status": "reported",
          "value": 1
        },
        "seed": {
          "notes": "Choice randomization seed is not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "No demonstrations.",
          "reporting_status": "reported",
          "value": "zero-shot"
        },
        "statistical": {
          "notes": "Precision equals accuracy and coverage is 100% in this no-refusal condition.",
          "reporting_status": "reported",
          "value": "accuracy, precision, and coverage over 205 forced-choice questions"
        },
        "system_prompt_public": {
          "notes": "Public MCQ-without-refusal prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Script default, not overridden.",
          "reporting_status": "reported",
          "value": 1.0
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "No browser.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "No code execution.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Direct model calls.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "No database tool.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "No capsule files or tools.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "No internet tool.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One forced-choice generation per question.",
          "reporting_status": "reported",
          "value": "single-turn model call"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bixbench-evidence-v1-5-mcq-no-refusal-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bixbench-gpt-4o-unversioned",
          "n": 205,
          "notes": "74 correct of 205.",
          "status": "verified",
          "value": 36.0975609756
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bixbench-evidence-v1-5-mcq-no-refusal-results"
          ],
          "metric_id": "precision",
          "model_id": "bixbench-gpt-4o-unversioned",
          "n": 205,
          "notes": "74 correct among 205 sure answers.",
          "status": "verified",
          "value": 36.0975609756
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bixbench-evidence-v1-5-mcq-no-refusal-results"
          ],
          "metric_id": "coverage",
          "model_id": "bixbench-gpt-4o-unversioned",
          "n": 205,
          "notes": "205 sure answers of 205.",
          "status": "verified",
          "value": 100.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bixbench-evidence-v1-5-mcq-no-refusal-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bixbench-claude-3-5-sonnet-unversioned",
          "n": 205,
          "notes": "70 correct of 205.",
          "status": "verified",
          "value": 34.1463414634
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bixbench-evidence-v1-5-mcq-no-refusal-results"
          ],
          "metric_id": "precision",
          "model_id": "bixbench-claude-3-5-sonnet-unversioned",
          "n": 205,
          "notes": "70 correct among 205 sure answers.",
          "status": "verified",
          "value": 34.1463414634
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bixbench-evidence-v1-5-mcq-no-refusal-results"
          ],
          "metric_id": "coverage",
          "model_id": "bixbench-claude-3-5-sonnet-unversioned",
          "n": 205,
          "notes": "205 sure answers of 205.",
          "status": "verified",
          "value": 100.0
        }
      ],
      "scope": {
        "filter": "All v1.5 rows, randomized substantive choices without a refusal option, no capsule files.",
        "n": 205,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact counts, values, no-refusal prompt, and deterministic grader verified.",
        "status": "verified"
      },
      "work_id": "bixbench-v1-5-release",
      "work_version_id": "bixbench-v1-5-release-2025-09-26"
    },
    {
      "benchmark_id": "bixbench",
      "benchmark_version": "v1.5",
      "comparability_group": "bixbench-v1-5-zero-shot-mcq-refusal-temperature-one",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bixbench-evidence-v1-5-mcq-refusal-protocol",
          "locator": {
            "note": "Defines full-set zero-shot MCQ calls, randomized options, refusal flag, temperature 1.0, and exact option grading.",
            "type": "repository-path",
            "value": "scripts/run_zeroshot.sh, generate_zeroshot_evals.py, bixbench/zero_shot.py, and grade_outputs.py at commit 49311180bdacb324c596f2e07596c126f2004008"
          },
          "source_id": "bixbench-repository-resource",
          "source_type": "resource",
          "supports": [
            "/benchmark_version",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bixbench-evidence-v1-5-mcq-refusal-results",
          "locator": {
            "note": "Exact accuracy, precision, coverage, n_total, n_correct, and n_sure for both refusal-condition baselines.",
            "type": "repository-path",
            "value": "bixbench-v1.5_results/zero_shot_baselines.json at commit 28909d842bc492ecd99bab303279afb29e3cb353"
          },
          "source_id": "bixbench-v1-5-results-resource",
          "source_type": "resource",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bixbench-v1-5-zero-shot-mcq-refusal",
      "metrics": [
        {
          "aggregation": "correct over all questions",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "accuracy",
          "tolerance": "exact normalized option",
          "unit": "percent"
        },
        {
          "aggregation": "correct over non-refusal answers",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "precision",
          "tolerance": "exact normalized option",
          "unit": "percent"
        },
        {
          "aggregation": "non-refusal answers over all questions",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "coverage",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "coverage",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bixbench-claude-3-5-sonnet-unversioned",
        "bixbench-gpt-4o-unversioned"
      ],
      "protocol": {
        "contamination": {
          "notes": "No model-specific decontamination analysis.",
          "reporting_status": "reported",
          "value": "public benchmark canary"
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "exact normalized selected-option match with a sure/refusal flag"
        },
        "reasoning": {
          "notes": "No explicit reasoning-effort parameter.",
          "reporting_status": "reported",
          "value": "default model behavior"
        },
        "repeats": {
          "notes": "One prediction per model and question.",
          "reporting_status": "reported",
          "value": 1
        },
        "seed": {
          "notes": "Choice randomization seed is not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "No demonstrations.",
          "reporting_status": "reported",
          "value": "zero-shot"
        },
        "statistical": {
          "notes": "No confidence intervals.",
          "reporting_status": "reported",
          "value": "accuracy, precision among non-refusals, and non-refusal coverage over 205 questions"
        },
        "system_prompt_public": {
          "notes": "Public MCQ-with-refusal prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Script default, not overridden.",
          "reporting_status": "reported",
          "value": 1.0
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "No browser.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "No code execution.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Direct model calls.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "No database tool.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "No capsule files or tools.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "No internet tool.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One choice generation per question.",
          "reporting_status": "reported",
          "value": "single-turn model call"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bixbench-evidence-v1-5-mcq-refusal-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bixbench-gpt-4o-unversioned",
          "n": 205,
          "notes": "8 correct of 205.",
          "status": "verified",
          "value": 3.9024390244
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bixbench-evidence-v1-5-mcq-refusal-results"
          ],
          "metric_id": "precision",
          "model_id": "bixbench-gpt-4o-unversioned",
          "n": 205,
          "notes": "8 correct among 20 non-refusals.",
          "status": "verified",
          "value": 40.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bixbench-evidence-v1-5-mcq-refusal-results"
          ],
          "metric_id": "coverage",
          "model_id": "bixbench-gpt-4o-unversioned",
          "n": 205,
          "notes": "20 non-refusals of 205.",
          "status": "verified",
          "value": 9.756097561
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bixbench-evidence-v1-5-mcq-refusal-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bixbench-claude-3-5-sonnet-unversioned",
          "n": 205,
          "notes": "17 correct of 205.",
          "status": "verified",
          "value": 8.2926829268
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bixbench-evidence-v1-5-mcq-refusal-results"
          ],
          "metric_id": "precision",
          "model_id": "bixbench-claude-3-5-sonnet-unversioned",
          "n": 205,
          "notes": "17 correct among 44 non-refusals.",
          "status": "verified",
          "value": 38.6363636364
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bixbench-evidence-v1-5-mcq-refusal-results"
          ],
          "metric_id": "coverage",
          "model_id": "bixbench-claude-3-5-sonnet-unversioned",
          "n": 205,
          "notes": "44 non-refusals of 205.",
          "status": "verified",
          "value": 21.4634146341
        }
      ],
      "scope": {
        "filter": "All v1.5 rows, randomized choices plus an Insufficient information option, no capsule files.",
        "n": 205,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact counts, values, prompt condition, and deterministic grader verified.",
        "status": "verified"
      },
      "work_id": "bixbench-v1-5-release",
      "work_version_id": "bixbench-v1-5-release-2025-09-26"
    },
    {
      "benchmark_id": "bixbench",
      "benchmark_version": "v1.5",
      "comparability_group": "bixbench-v1-5-zero-shot-open-temperature-one",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "bixbench-evidence-v1-5-open-protocol",
          "locator": {
            "note": "Defines full-dataset single calls, temperature 1.0, public prompts, no tools/files, and mixed open-answer verifiers.",
            "type": "repository-path",
            "value": "generate_zeroshot_evals.py, grade_outputs.py, bixbench/zero_shot.py, bixbench/graders.py, and scripts/run_zeroshot.sh at commit 49311180bdacb324c596f2e07596c126f2004008"
          },
          "source_id": "bixbench-repository-resource",
          "source_type": "resource",
          "supports": [
            "/benchmark_version",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "bixbench-evidence-v1-5-open-results",
          "locator": {
            "note": "Both models have n_total 205, n_correct 6, n_sure 205.",
            "type": "repository-path",
            "value": "bixbench-v1.5_results/zero_shot_baselines.json at commit 28909d842bc492ecd99bab303279afb29e3cb353"
          },
          "source_id": "bixbench-v1-5-results-resource",
          "source_type": "resource",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "bixbench-v1-5-zero-shot-open",
      "metrics": [
        {
          "aggregation": "correct over all 205 questions",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "accuracy",
          "tolerance": "verifier-specific",
          "unit": "percent"
        },
        {
          "aggregation": "correct over sure predictions",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "precision",
          "tolerance": "verifier-specific",
          "unit": "percent"
        },
        {
          "aggregation": "sure predictions over all 205 questions",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "coverage",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "coverage",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "bixbench-claude-3-5-sonnet-unversioned",
        "bixbench-gpt-4o-unversioned"
      ],
      "protocol": {
        "contamination": {
          "notes": "The dataset includes a canary and no-training warning; no model-specific decontamination analysis.",
          "reporting_status": "reported",
          "value": "public benchmark canary"
        },
        "grader": {
          "human_review": false,
          "model": "GPT-4o for llm_verifier rows (exact endpoint not reported)",
          "reporting_status": "reported",
          "type": "row-specific string, numeric-range, or LLM verifier"
        },
        "reasoning": {
          "notes": "No explicit reasoning-effort parameter.",
          "reporting_status": "reported",
          "value": "default model behavior"
        },
        "repeats": {
          "notes": "The release JSON contains one prediction per model and question.",
          "reporting_status": "reported",
          "value": 1
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The prompt contains the question but no demonstrations or capsule files.",
          "reporting_status": "reported",
          "value": "zero-shot"
        },
        "statistical": {
          "notes": "No confidence intervals.",
          "reporting_status": "reported",
          "value": "accuracy over 205 questions, plus precision and coverage from correct/sure flags"
        },
        "system_prompt_public": {
          "notes": "The open-ended prompt template is public in bixbench/prompts.py.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Default in generate_zeroshot_evals.py; the release script does not override it.",
          "reporting_status": "reported",
          "value": 1.0
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "No browser.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "No notebook or code execution.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Direct model calls rather than agent containers.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "No database tool.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "No capsule files or external tools.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "No internet tool.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One generation per question.",
          "reporting_status": "reported",
          "value": "single-turn model call"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bixbench-evidence-v1-5-open-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bixbench-gpt-4o-unversioned",
          "n": 205,
          "notes": "6 correct of 205.",
          "status": "verified",
          "value": 2.9268292683
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bixbench-evidence-v1-5-open-results"
          ],
          "metric_id": "precision",
          "model_id": "bixbench-gpt-4o-unversioned",
          "n": 205,
          "notes": "6 correct among 205 sure predictions.",
          "status": "verified",
          "value": 2.9268292683
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bixbench-evidence-v1-5-open-results"
          ],
          "metric_id": "coverage",
          "model_id": "bixbench-gpt-4o-unversioned",
          "n": 205,
          "notes": "205 sure predictions of 205.",
          "status": "verified",
          "value": 100.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bixbench-evidence-v1-5-open-results"
          ],
          "metric_id": "accuracy",
          "model_id": "bixbench-claude-3-5-sonnet-unversioned",
          "n": 205,
          "notes": "6 correct of 205.",
          "status": "verified",
          "value": 2.9268292683
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bixbench-evidence-v1-5-open-results"
          ],
          "metric_id": "precision",
          "model_id": "bixbench-claude-3-5-sonnet-unversioned",
          "n": 205,
          "notes": "6 correct among 205 sure predictions.",
          "status": "verified",
          "value": 2.9268292683
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "bixbench-evidence-v1-5-open-results"
          ],
          "metric_id": "coverage",
          "model_id": "bixbench-claude-3-5-sonnet-unversioned",
          "n": 205,
          "notes": "205 sure predictions of 205.",
          "status": "verified",
          "value": 100.0
        }
      ],
      "scope": {
        "filter": "All v1.5 rows, question text only, open-answer prompt.",
        "n": 205,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact JSON values and public generation/grading code verified at pinned creator commits.",
        "status": "verified"
      },
      "work_id": "bixbench-v1-5-release",
      "work_version_id": "bixbench-v1-5-release-2025-09-26"
    },
    {
      "benchmark_id": "blade-mcq",
      "benchmark_version": "arXiv v3",
      "comparability_group": "blade-arxiv-v3-mcq-zero-shot-temperature-0",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "blade-evidence-mcq-protocol",
          "locator": {
            "note": "Reports 188 MCQs, nine evaluated model families, temperature zero, accuracy with 95% intervals, and the public prompt.",
            "type": "section",
            "value": "arXiv v3 §§4.1 and 6, Figure 3, and Appendix A.6 Figure 19"
          },
          "source_id": "blade-mcq-paper-resource",
          "source_type": "resource",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol/temperature",
            "/protocol/statistical",
            "/metrics",
            "/comparability_group"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "blade-evidence-mcq-code",
          "locator": {
            "note": "Confirms one direct prompt per item, no tool loop, exact-choice grading, complete public prompt, 20/168 question components, and available exact model strings.",
            "type": "repository-path",
            "value": "run_mcq.py, run_scripts/sh_run_mcq.sh, blade_bench/baselines/lm/mcq.py, blade_bench/eval/datamodel/run_mcq.py, and blade_bench/conf/llm_config.yml at commit 6118fa8d5007b91aa8c91c518182db82446a4547"
          },
          "source_id": "blade-mcq-repository-resource",
          "source_type": "resource",
          "supports": [
            "/scope",
            "/protocol/shots",
            "/protocol/turns",
            "/protocol/system_prompt_public",
            "/protocol/reasoning",
            "/protocol/tools",
            "/protocol/repeats",
            "/protocol/grader",
            "/model_ids"
          ]
        }
      ],
      "id": "blade-creator-decision-mcq",
      "metrics": [
        {
          "aggregation": "correct choices divided by all 188 MCQs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Accuracy",
          "tolerance": "exact option identifier",
          "unit": "percent"
        }
      ],
      "model_ids": [
        "blade-claude-3-5-sonnet-20240620",
        "blade-codellama-7b-instruct",
        "blade-deepseek-coder-6-7b-instruct",
        "blade-gemini-1-5-pro-unversioned",
        "blade-gpt35-turbo-unversioned",
        "blade-gpt4o-unversioned",
        "blade-llama3-70b-unversioned",
        "blade-llama3-8b-unversioned",
        "blade-mixtral-8x22b-unversioned"
      ],
      "protocol": {
        "contamination": {
          "notes": "No pretraining-contamination or decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "deterministic exact option match"
        },
        "reasoning": {
          "notes": "No provider reasoning-effort control is reported.",
          "reporting_status": "reported",
          "value": "direct multiple-choice completion"
        },
        "repeats": {
          "notes": "The public runner iterates once over each MCQ; the paper reports no repeated MCQ sampling.",
          "reporting_status": "reported",
          "value": 1
        },
        "seed": {
          "notes": "No seed is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The public MCQ prompt contains no in-context solved example.",
          "reporting_status": "reported",
          "value": "zero-shot"
        },
        "statistical": {
          "notes": "The paper does not state the confidence-interval construction; numeric Figure 3 values are not printed in a table and are therefore not digitized here.",
          "reporting_status": "reported",
          "value": "question-weighted accuracy with 95% confidence intervals"
        },
        "system_prompt_public": {
          "notes": "The system and instruction prompt are public in the repository and paper appendix.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "The creator paper explicitly sets temperature zero for Task 1.",
          "reporting_status": "reported",
          "value": 0
        },
        "time_budget": {
          "notes": "No wall-clock limit is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No per-question token limit is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "No browser tool is exposed.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "The MCQ runner requests a direct response without code execution.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No interactive sandbox is exposed.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "No external database tool is exposed.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "The model receives the research question, serialized dataset description, and choices.",
            "reporting_status": "reported",
            "value": "none"
          },
          "internet": {
            "notes": "No network tool is exposed.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One completion returns a selected option and rationale for each question.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [],
      "scope": {
        "filter": "All 20 conceptual-variable and 168 transformation MCQs.",
        "n": 188,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Full 188-question scope, zero-shot single-turn prompt, temperature, no-tool setting, deterministic grader, and metric were verified. Figure-only values are intentionally not digitized.",
        "status": "verified"
      },
      "work_id": "blade-paper",
      "work_version_id": "blade-paper-2024-11-16"
    },
    {
      "benchmark_id": "blade-analysis-generation",
      "benchmark_version": "arXiv v3",
      "comparability_group": "blade-arxiv-v3-generation-one-turn-40-temperature-08",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "blade-evidence-one-turn-protocol",
          "locator": {
            "note": "Defines the three submitted artifacts, one-shot one-turn prompt, temperature 0.8, 40 runs, GPT-4o-assisted evaluation, average precision, coverage@10, weighted F1, and 1,000-iteration bootstrap.",
            "type": "section",
            "value": "arXiv v3 §§4.2, 5–6 and Appendices A.6–A.7"
          },
          "source_id": "blade-generation-paper-resource",
          "source_type": "resource",
          "supports": [
            "/benchmark_version",
            "/scope",
            "/protocol",
            "/metrics",
            "/comparability_group"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "blade-evidence-one-turn-results",
          "locator": {
            "note": "Prints all nine decision-type weighted F1 point estimates and 95% confidence intervals.",
            "type": "table",
            "value": "Table 2, One-turn Setting"
          },
          "source_id": "blade-generation-paper-resource",
          "source_type": "resource",
          "supports": [
            "/model_ids",
            "/results"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "blade-evidence-one-turn-code",
          "locator": {
            "note": "Confirms all 12 datasets, 40 runs, direct generation, public prompt/evaluator, and the available exact model strings.",
            "type": "repository-path",
            "value": "run_scripts/sh_run_one_turn.sh, blade_bench/baselines/run.py, blade_bench/baselines/lm/analysis.py, blade_bench/conf/llm_config.yml, and blade_bench/eval/ at commit 6118fa8d5007b91aa8c91c518182db82446a4547"
          },
          "source_id": "blade-generation-repository-resource",
          "source_type": "resource",
          "supports": [
            "/model_ids",
            "/protocol/shots",
            "/protocol/turns",
            "/protocol/system_prompt_public",
            "/protocol/tools",
            "/protocol/repeats",
            "/protocol/grader"
          ]
        }
      ],
      "id": "blade-creator-paper",
      "metrics": [
        {
          "aggregation": "mean of per-run precision followed by macro averaging across datasets for each decision type",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "average-precision",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Average precision",
          "tolerance": "GPT-4o semantic matching plus transformation value/graph matching",
          "unit": "percent"
        },
        {
          "aggregation": "ground-truth coverage of the union of ten sampled runs followed by macro averaging across datasets for each decision type",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "coverage-at-10",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Coverage@10",
          "tolerance": "run errors count as zero-coverage generations",
          "unit": "percent"
        },
        {
          "aggregation": "weighted mean of per-decision-type harmonic means of average precision and coverage@10",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "decision-weighted-f1",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Decision-type weighted F1-score",
          "tolerance": "1000-run bootstrap mean and 95% interval",
          "unit": "percent"
        }
      ],
      "model_ids": [
        "blade-claude-3-5-sonnet-20240620",
        "blade-codellama-7b-instruct",
        "blade-deepseek-coder-6-7b-instruct",
        "blade-gemini-1-5-pro-unversioned",
        "blade-gpt35-turbo-unversioned",
        "blade-gpt4o-unversioned",
        "blade-llama3-70b-unversioned",
        "blade-llama3-8b-unversioned",
        "blade-mixtral-8x22b-unversioned"
      ],
      "protocol": {
        "contamination": {
          "notes": "No pretraining-contamination or decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": "GPT-4o (exact snapshot not reported)",
          "reporting_status": "reported",
          "type": "hybrid automatic executable and decision matching"
        },
        "reasoning": {
          "notes": "No provider reasoning-effort control is reported.",
          "reporting_status": "reported",
          "value": "direct prompted generation without an agent loop"
        },
        "repeats": {
          "notes": "Forty independent generations per model and source dataset.",
          "reporting_status": "reported",
          "value": 40
        },
        "seed": {
          "notes": "No random seed is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The prompt includes one complete worked analysis example.",
          "reporting_status": "reported",
          "value": "one-shot"
        },
        "statistical": {
          "notes": "Run errors receive hit rate zero and remain in coverage. Modeling-decision weighting uses min(|G_model|, 10).",
          "reporting_status": "reported",
          "value": "For each dataset and decision type, average precision across all runs and coverage@10 feed a harmonic-mean F1; decision-type F1 values are weighted by ground-truth decision counts. Reported F1 is the mean of 1,000 bootstrap resamples with a percentile-derived 95% interval."
        },
        "system_prompt_public": {
          "notes": "The full system, task, schema, format, and worked-example prompts are published in the appendix and repository.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Used to encourage diverse valid analysis decisions across repeated runs.",
          "reporting_status": "reported",
          "value": 0.8
        },
        "time_budget": {
          "notes": "No wall-clock limit is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No per-run token limit is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "No browser tool is exposed in the one-turn baseline.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "The model does not execute code during generation; the evaluator subsequently executes the submitted transform/model code.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No interactive sandbox is exposed to the one-turn model.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "No external database tool is exposed.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "The model receives the question and serialized dataset schema directly.",
            "reporting_status": "reported",
            "value": "none"
          },
          "internet": {
            "notes": "No network tool is exposed in the one-turn baseline.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One model completion produces the conceptual variables, transform function, and statistical-model function.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": 18.5,
          "ci_low": 15.2,
          "confidence": "high",
          "evidence_ids": [
            "blade-evidence-one-turn-results"
          ],
          "metric_id": "decision-weighted-f1",
          "model_id": "blade-codellama-7b-instruct",
          "n": 12,
          "notes": "Forty one-turn generations per dataset; Table 2.",
          "status": "verified",
          "value": 16.8
        },
        {
          "ci_high": 35.4,
          "ci_low": 32.2,
          "confidence": "high",
          "evidence_ids": [
            "blade-evidence-one-turn-results"
          ],
          "metric_id": "decision-weighted-f1",
          "model_id": "blade-deepseek-coder-6-7b-instruct",
          "n": 12,
          "notes": "Forty one-turn generations per dataset; Table 2.",
          "status": "verified",
          "value": 33.9
        },
        {
          "ci_high": 31.5,
          "ci_low": 27.7,
          "confidence": "high",
          "evidence_ids": [
            "blade-evidence-one-turn-results"
          ],
          "metric_id": "decision-weighted-f1",
          "model_id": "blade-llama3-8b-unversioned",
          "n": 12,
          "notes": "Forty one-turn generations per dataset; exact served snapshot not reported.",
          "status": "verified",
          "value": 29.6
        },
        {
          "ci_high": 37.8,
          "ci_low": 34.7,
          "confidence": "high",
          "evidence_ids": [
            "blade-evidence-one-turn-results"
          ],
          "metric_id": "decision-weighted-f1",
          "model_id": "blade-llama3-70b-unversioned",
          "n": 12,
          "notes": "Forty one-turn generations per dataset; exact served snapshot not reported.",
          "status": "verified",
          "value": 36.3
        },
        {
          "ci_high": 42.1,
          "ci_low": 38.0,
          "confidence": "high",
          "evidence_ids": [
            "blade-evidence-one-turn-results"
          ],
          "metric_id": "decision-weighted-f1",
          "model_id": "blade-mixtral-8x22b-unversioned",
          "n": 12,
          "notes": "Forty one-turn generations per dataset; exact served snapshot not reported.",
          "status": "verified",
          "value": 40.1
        },
        {
          "ci_high": 32.2,
          "ci_low": 28.7,
          "confidence": "high",
          "evidence_ids": [
            "blade-evidence-one-turn-results"
          ],
          "metric_id": "decision-weighted-f1",
          "model_id": "blade-gpt35-turbo-unversioned",
          "n": 12,
          "notes": "Forty one-turn generations per dataset; dated endpoint snapshot not reported.",
          "status": "verified",
          "value": 30.5
        },
        {
          "ci_high": 43.2,
          "ci_low": 40.2,
          "confidence": "high",
          "evidence_ids": [
            "blade-evidence-one-turn-results"
          ],
          "metric_id": "decision-weighted-f1",
          "model_id": "blade-gpt4o-unversioned",
          "n": 12,
          "notes": "Forty one-turn generations per dataset; dated endpoint snapshot not reported.",
          "status": "verified",
          "value": 41.7
        },
        {
          "ci_high": 42.5,
          "ci_low": 39.6,
          "confidence": "high",
          "evidence_ids": [
            "blade-evidence-one-turn-results"
          ],
          "metric_id": "decision-weighted-f1",
          "model_id": "blade-gemini-1-5-pro-unversioned",
          "n": 12,
          "notes": "Forty one-turn generations per dataset; immutable endpoint snapshot not reported.",
          "status": "verified",
          "value": 41.1
        },
        {
          "ci_high": 44.9,
          "ci_low": 42.6,
          "confidence": "high",
          "evidence_ids": [
            "blade-evidence-one-turn-results"
          ],
          "metric_id": "decision-weighted-f1",
          "model_id": "blade-claude-3-5-sonnet-20240620",
          "n": 12,
          "notes": "Forty one-turn generations per dataset; Table 2.",
          "status": "verified",
          "value": 43.9
        }
      ],
      "scope": {
        "filter": "All twelve BLADE research-question/dataset pairs in the one-turn analysis-generation setting.",
        "n": 12,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Full scope, one-shot/single-turn setting, no-tool baseline, 40 repeats, temperature, hybrid grader, metric construction, bootstrap, and all nine Table 2 results were verified.",
        "status": "verified"
      },
      "work_id": "blade-paper",
      "work_version_id": "blade-paper-2024-11-16"
    },
    {
      "benchmark_id": "blade-analysis-generation",
      "benchmark_version": "arXiv v3",
      "comparability_group": "blade-arxiv-v3-generation-react-20-temperature-08",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "blade-evidence-react-protocol",
          "locator": {
            "note": "Defines the ReAct notebook, one trajectory example, Python 3.10 environment, ten steps, temperature 0.8, 20 runs, GPT-4o-assisted evaluation, metrics, and 1,000-iteration bootstrap.",
            "type": "section",
            "value": "arXiv v3 §§5–6 and Appendices A.5–A.7"
          },
          "source_id": "blade-generation-paper-resource",
          "source_type": "resource",
          "supports": [
            "/benchmark_version",
            "/scope",
            "/protocol",
            "/metrics",
            "/comparability_group"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "blade-evidence-react-results",
          "locator": {
            "note": "Prints all five ReAct decision-type weighted F1 point estimates and 95% confidence intervals.",
            "type": "table",
            "value": "Table 2, Agent Setting"
          },
          "source_id": "blade-generation-paper-resource",
          "source_type": "resource",
          "supports": [
            "/model_ids",
            "/results"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "blade-evidence-react-code",
          "locator": {
            "note": "Confirms all 12 datasets, 20 runs, ten-step ReAct execution, public prompt/evaluator, and available exact model strings.",
            "type": "repository-path",
            "value": "run_scripts/sh_run_agent.sh, blade_bench/baselines/agent/react_agent.py, blade_bench/baselines/run.py, blade_bench/conf/llm_config.yml, and blade_bench/eval/ at commit 6118fa8d5007b91aa8c91c518182db82446a4547"
          },
          "source_id": "blade-generation-repository-resource",
          "source_type": "resource",
          "supports": [
            "/model_ids",
            "/protocol/shots",
            "/protocol/turns",
            "/protocol/system_prompt_public",
            "/protocol/reasoning",
            "/protocol/tools",
            "/protocol/time_budget",
            "/protocol/repeats",
            "/protocol/grader"
          ]
        }
      ],
      "id": "blade-creator-react",
      "metrics": [
        {
          "aggregation": "mean of per-run precision followed by macro averaging across datasets for each decision type",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "average-precision",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Average precision",
          "tolerance": "GPT-4o semantic matching plus transformation value/graph matching",
          "unit": "percent"
        },
        {
          "aggregation": "ground-truth coverage of the union of ten sampled runs followed by macro averaging across datasets for each decision type",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "coverage-at-10",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Coverage@10",
          "tolerance": "run errors count as zero-coverage generations",
          "unit": "percent"
        },
        {
          "aggregation": "weighted mean of per-decision-type harmonic means of average precision and coverage@10",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "decision-weighted-f1",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Decision-type weighted F1-score",
          "tolerance": "1000-run bootstrap mean and 95% interval",
          "unit": "percent"
        }
      ],
      "model_ids": [
        "blade-claude-3-5-sonnet-20240620",
        "blade-gemini-1-5-pro-unversioned",
        "blade-gpt35-turbo-unversioned",
        "blade-gpt4o-unversioned",
        "blade-mixtral-8x22b-unversioned"
      ],
      "protocol": {
        "contamination": {
          "notes": "No pretraining-contamination or decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": "GPT-4o (exact snapshot not reported)",
          "reporting_status": "reported",
          "type": "hybrid automatic executable and decision matching"
        },
        "reasoning": {
          "notes": "The model can finish early; otherwise a final completion is requested after ten steps.",
          "reporting_status": "reported",
          "value": "ReAct loop with full prior thought, action, and observation context"
        },
        "repeats": {
          "notes": "Twenty independent agent trajectories per model and source dataset.",
          "reporting_status": "reported",
          "value": 20
        },
        "seed": {
          "notes": "No random seed is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The public prompt includes one Thought-Action-Observation example.",
          "reporting_status": "reported",
          "value": "one-shot ReAct trajectory"
        },
        "statistical": {
          "notes": "Run errors receive hit rate zero and remain in coverage. Modeling-decision weighting uses min(|G_model|, 10).",
          "reporting_status": "reported",
          "value": "For each dataset and decision type, average precision across all runs and coverage@10 feed a harmonic-mean F1; decision-type F1 values are weighted by ground-truth decision counts. Reported F1 is the mean of 1,000 bootstrap resamples with a percentile-derived 95% interval."
        },
        "system_prompt_public": {
          "notes": "The full ReAct and output-format prompts are published in the appendix and repository.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Used to encourage diverse valid analysis decisions across repeated runs.",
          "reporting_status": "reported",
          "value": 0.8
        },
        "time_budget": {
          "notes": "No wall-clock limit is reported.",
          "reporting_status": "reported",
          "value": "maximum 10 agent steps"
        },
        "token_budget": {
          "notes": "No per-run token limit is reported; candidate models needed at least an 8k context window.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "The agent exposes only the computational notebook action.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Each Action executes a new Python cell and returns the last-line output as the Observation.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "The paper describes a sandbox notebook using Python 3.10 with pandas, sklearn, scipy, statsmodels, NumPy, Matplotlib, and seaborn.",
            "reporting_status": "reported",
            "value": true
          },
          "databases": {
            "notes": "No external database tool is exposed.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "The underlying dataset is mounted in the notebook environment.",
            "reporting_status": "reported",
            "value": "computational notebook cell execution"
          },
          "internet": {
            "notes": "No network tool is described or exposed by the published agent.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "At most ten Thought-Action-Observation steps followed by a final analysis.",
          "reporting_status": "reported",
          "value": "multi-turn"
        }
      },
      "results": [
        {
          "ci_high": 42.9,
          "ci_low": 38.2,
          "confidence": "high",
          "evidence_ids": [
            "blade-evidence-react-results"
          ],
          "metric_id": "decision-weighted-f1",
          "model_id": "blade-mixtral-8x22b-unversioned",
          "n": 12,
          "notes": "Twenty ReAct trajectories per dataset; exact served snapshot not reported.",
          "status": "verified",
          "value": 40.8
        },
        {
          "ci_high": 39.7,
          "ci_low": 34.7,
          "confidence": "high",
          "evidence_ids": [
            "blade-evidence-react-results"
          ],
          "metric_id": "decision-weighted-f1",
          "model_id": "blade-gpt35-turbo-unversioned",
          "n": 12,
          "notes": "Twenty ReAct trajectories per dataset; dated endpoint snapshot not reported.",
          "status": "verified",
          "value": 37.2
        },
        {
          "ci_high": 46.3,
          "ci_low": 43.0,
          "confidence": "high",
          "evidence_ids": [
            "blade-evidence-react-results"
          ],
          "metric_id": "decision-weighted-f1",
          "model_id": "blade-gpt4o-unversioned",
          "n": 12,
          "notes": "Twenty ReAct trajectories per dataset; dated endpoint snapshot not reported.",
          "status": "verified",
          "value": 44.8
        },
        {
          "ci_high": 41.5,
          "ci_low": 38.3,
          "confidence": "high",
          "evidence_ids": [
            "blade-evidence-react-results"
          ],
          "metric_id": "decision-weighted-f1",
          "model_id": "blade-gemini-1-5-pro-unversioned",
          "n": 12,
          "notes": "Twenty ReAct trajectories per dataset; immutable endpoint snapshot not reported.",
          "status": "verified",
          "value": 40.1
        },
        {
          "ci_high": 44.8,
          "ci_low": 41.4,
          "confidence": "high",
          "evidence_ids": [
            "blade-evidence-react-results"
          ],
          "metric_id": "decision-weighted-f1",
          "model_id": "blade-claude-3-5-sonnet-20240620",
          "n": 12,
          "notes": "Twenty ReAct trajectories per dataset; Table 2.",
          "status": "verified",
          "value": 43.1
        }
      ],
      "scope": {
        "filter": "All twelve BLADE research-question/dataset pairs in the ReAct notebook-agent setting.",
        "n": 12,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Full scope, one-shot ReAct setting, ten-step Python notebook, 20 repeats, temperature, hybrid grader, metric construction, bootstrap, and all five Table 2 results were verified.",
        "status": "verified"
      },
      "work_id": "blade-paper",
      "work_version_id": "blade-paper-2024-11-16"
    },
    {
      "benchmark_id": "cameo",
      "benchmark_version": "2024-complex-study",
      "comparability_group": "cameo-2024-antibody-three-server-common",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "cameo-2024-antibody-result-evidence",
          "locator": {
            "note": "Reports 83 antibody common-subset targets, AF3 median LDDT 0.83, MultiFOLD median LDDT 0.76, and the model-1 comparison protocol.",
            "type": "section",
            "value": "Section 2.4.2 and Figure 2A"
          },
          "source_id": "cameo-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics",
            "/results"
          ]
        }
      ],
      "id": "cameo-2024-antibody-three-server-common",
      "metrics": [
        {
          "aggregation": "median across 83 antibody common-subset targets",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "median-lddt",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Median Complex LDDT",
          "tolerance": "missing chains are penalized",
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "cameo-alphafold3-v301",
        "cameo-multifold-2024",
        "cameo-swissmodel-2024"
      ],
      "protocol": {
        "contamination": {
          "notes": "Stoichiometry is not supplied in the PDB pre-release.",
          "reporting_status": "reported",
          "value": "experimental antibody-complex structures withheld until PDB release"
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "fully automated OpenStructure complex LDDT scoring"
        },
        "reasoning": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "repeats": {
          "notes": "Creator analysis uses the top-ranked/model-1 prediction.",
          "reporting_status": "reported",
          "value": "up to five submitted models per target"
        },
        "seed": {
          "notes": "Method-specific stochastic settings are not normalized.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "statistical": {
          "notes": "Sixty-one of 83 targets contain one protein chain per entity; the paper reports medians for AlphaFold 3 and MultiFOLD.",
          "reporting_status": "reported",
          "value": "median Complex LDDT across 83 common-subset antibody targets"
        },
        "system_prompt_public": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "temperature": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "time_budget": {
          "notes": "Saturday post-pre-release submission through Wednesday 00:00 UTC.",
          "reporting_status": "reported",
          "value": "approximately 3.5 days per weekly target"
        },
        "token_budget": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "No common interactive-browser setting applies.",
            "reporting_status": "reported",
            "value": "method-specific"
          },
          "code_execution": {
            "notes": "Registered server pipelines execute prediction code.",
            "reporting_status": "reported",
            "value": "allowed"
          },
          "container": {
            "notes": "No common container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "SAbDab is used post hoc to classify targets; experimental targets remain blind during prediction.",
            "reporting_status": "reported",
            "value": "method-specific public structure and sequence data"
          },
          "external_tools": {
            "notes": "AlphaFold 3 v3.0.1, MultiFOLD, and SWISS-MODEL.",
            "reporting_status": "reported",
            "value": "method-specific server pipeline"
          },
          "internet": {
            "notes": "All servers use the same weekly public-information cutoff.",
            "reporting_status": "reported",
            "value": "allowed before each weekly deadline"
          }
        },
        "turns": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "cameo-2024-antibody-result-evidence"
          ],
          "metric_id": "median-lddt",
          "model_id": "cameo-alphafold3-v301",
          "n": 83,
          "notes": "Median reported in creator-paper Section 2.4.2.",
          "status": "verified",
          "value": 0.83
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "cameo-2024-antibody-result-evidence"
          ],
          "metric_id": "median-lddt",
          "model_id": "cameo-multifold-2024",
          "n": 83,
          "notes": "Median reported in creator-paper Section 2.4.2.",
          "status": "verified",
          "value": 0.76
        }
      ],
      "scope": {
        "filter": "SAbDab-mapped antibody targets in the AlphaFold 3, MultiFOLD, and SWISS-MODEL common subset.",
        "n": 83,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "cameo-2024-antibody-three-server-common",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Only two explicitly reported medians are ingested; no value is inferred for SWISS-MODEL from the plot.",
        "status": "verified"
      },
      "work_id": "cameo-paper",
      "work_version_id": "cameo-paper-2025-09-28"
    },
    {
      "benchmark_id": "cameo",
      "benchmark_version": "2024-complex-study",
      "comparability_group": "cameo-2024-ligand-four-baseline-common",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "cameo-2024-ligand-run-evidence",
          "locator": {
            "note": "Reports the four baselines, 2,584-target common subset, 6,152 entities, top-ranked model handling, ligand metrics, and exact pipeline versions.",
            "type": "section",
            "value": "Sections 2.3, 2.4.1, 2.5.1-2.5.2, and 3.2; Figure 1"
          },
          "source_id": "cameo-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        }
      ],
      "id": "cameo-2024-ligand-baseline-common",
      "metrics": [
        {
          "aggregation": "percentage across common-subset ligand entities",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "ligand-success-rate",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Ligand success rate",
          "tolerance": "A success is a symmetry-corrected BiSyRMSD below 2 Å after binding-site superposition.",
          "unit": "percent of ligand entities"
        },
        {
          "aggregation": "weighted aggregation across relevant ligand entities",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "lddt-pli",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "LDDT-PLI",
          "tolerance": null,
          "unit": "proportion"
        },
        {
          "aggregation": "best-scored submitted pose per ligand entity in the creator analysis",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "bisyrmsd",
          "pass_threshold": 2,
          "range": null,
          "source_label": "BiSyRMSD",
          "tolerance": "symmetry-corrected after binding-site superposition",
          "unit": "angstrom"
        }
      ],
      "model_ids": [
        "cameo-alphafold3-v301",
        "cameo-swissmodel-glide",
        "cameo-swissmodel-vina-ad4",
        "cameo-swissmodel-vina-vina"
      ],
      "protocol": {
        "contamination": {
          "notes": "All baselines receive the same weekly information cutoff; AlphaFold 3 training-date similarity is analyzed separately.",
          "reporting_status": "reported",
          "value": "PDB pre-release sequences and ligand identities available while experimental structures and poses remain withheld"
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "fully automated OpenStructure complex and ligand scoring"
        },
        "reasoning": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "repeats": {
          "notes": "The paper analyzes the top-ranked/model-1 prediction and selects the best-scored pose per ligand entity.",
          "reporting_status": "reported",
          "value": "up to five models or ligand poses per target"
        },
        "seed": {
          "notes": "Method-specific stochastic settings are not normalized.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "statistical": {
          "notes": "Ligand scores are aggregated across relevant non-polymer entities; common crystallographic artifacts are analyzed separately.",
          "reporting_status": "reported",
          "value": "common-subset aggregation across targets predicted by all four baselines"
        },
        "system_prompt_public": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "temperature": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "time_budget": {
          "notes": "Saturday post-pre-release submission through Wednesday 00:00 UTC; a short email grace period is documented by the current service.",
          "reporting_status": "reported",
          "value": "approximately 3.5 days per weekly target"
        },
        "token_budget": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "No common interactive-browser setting applies to registered structure-prediction servers.",
            "reporting_status": "reported",
            "value": "method-specific"
          },
          "code_execution": {
            "notes": "Each baseline executes its documented modeling and docking pipeline.",
            "reporting_status": "reported",
            "value": "allowed"
          },
          "container": {
            "notes": "No common container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Reference complex structures are withheld until PDB release.",
            "reporting_status": "reported",
            "value": "method-specific public structure and sequence data"
          },
          "external_tools": {
            "notes": "The two Vina baselines differ only by vina versus AutoDock4 scoring functions; no pocket predictor is used.",
            "reporting_status": "reported",
            "value": "AlphaFold 3 v3.0.1 or SWISS-MODEL followed by Glide or AutoDock Vina 1.2.5"
          },
          "internet": {
            "notes": "All servers are evaluated simultaneously with the same public sequence and template background date.",
            "reporting_status": "reported",
            "value": "allowed before each weekly deadline"
          }
        },
        "turns": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        }
      },
      "results": [],
      "scope": {
        "filter": "Targets in the 2024 ligand class predicted by all four baseline servers; 6,152 non-polymer entities.",
        "n": 2584,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "cameo-2024-ligand-baseline-common",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact baseline systems and protocol are registered; Figure 1 distributions are not transcribed into unsupported scalar result rows.",
        "status": "verified"
      },
      "work_id": "cameo-paper",
      "work_version_id": "cameo-paper-2025-09-28"
    },
    {
      "benchmark_id": "cameo",
      "benchmark_version": "2024-complex-study",
      "comparability_group": "cameo-2024-ppi-three-server-common",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "cameo-2024-ppi-run-evidence",
          "locator": {
            "note": "Reports the 392-target common subset, three systems, blind stoichiometry setting, model-1 analysis, and four metrics.",
            "type": "section",
            "value": "Sections 2.2, 2.4.2, and 2.5.1; Figure 2"
          },
          "source_id": "cameo-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        }
      ],
      "id": "cameo-2024-ppi-three-server-common",
      "metrics": [
        {
          "aggregation": "per target distribution with penalties for missing chains",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "lddt",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Complex LDDT",
          "tolerance": null,
          "unit": "proportion"
        },
        {
          "aggregation": "per target distribution over mapped chains",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "mapped-lddt",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Mapped complex LDDT",
          "tolerance": "removes the missing-chain stoichiometry penalty",
          "unit": "proportion"
        },
        {
          "aggregation": "per target inter-chain-contact distribution with stoichiometry penalties",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "interface-lddt",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Complex iLDDT",
          "tolerance": null,
          "unit": "proportion"
        },
        {
          "aggregation": "per target inter-chain contacts over mapped chains",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "mapped-interface-lddt",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Mapped complex iLDDT",
          "tolerance": "removes the missing-chain stoichiometry penalty",
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "cameo-alphafold3-v301",
        "cameo-multifold-2024",
        "cameo-swissmodel-2024"
      ],
      "protocol": {
        "contamination": {
          "notes": "Stoichiometry is absent from the PDB pre-release and must be predicted.",
          "reporting_status": "reported",
          "value": "experimental complex structures withheld until the weekly PDB release"
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "fully automated OpenStructure whole-complex and interface scoring"
        },
        "reasoning": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "repeats": {
          "notes": "Only the top-ranked/model-1 prediction is used in the creator-paper analysis.",
          "reporting_status": "reported",
          "value": "up to five submitted models per target"
        },
        "seed": {
          "notes": "Method-specific stochastic settings are not normalized.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "statistical": {
          "notes": "Results are stratified into antibody, homomer, and non-antibody heteromer classes; no single ranking is defined.",
          "reporting_status": "reported",
          "value": "score distributions on the three-server common subset"
        },
        "system_prompt_public": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "temperature": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "time_budget": {
          "notes": "Saturday post-pre-release submission through Wednesday 00:00 UTC.",
          "reporting_status": "reported",
          "value": "approximately 3.5 days per weekly target"
        },
        "token_budget": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "No common interactive-browser setting applies.",
            "reporting_status": "reported",
            "value": "method-specific"
          },
          "code_execution": {
            "notes": "Registered server pipelines execute their own prediction code.",
            "reporting_status": "reported",
            "value": "allowed"
          },
          "container": {
            "notes": "No common container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "The experimental complex structure remains withheld.",
            "reporting_status": "reported",
            "value": "method-specific public structure and sequence data"
          },
          "external_tools": {
            "notes": "The work compares AlphaFold 3 v3.0.1, MultiFOLD, and SWISS-MODEL.",
            "reporting_status": "reported",
            "value": "method-specific server pipeline"
          },
          "internet": {
            "notes": "All servers are evaluated simultaneously with the same public information cutoff.",
            "reporting_status": "reported",
            "value": "allowed before each weekly deadline"
          }
        },
        "turns": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        }
      },
      "results": [],
      "scope": {
        "filter": "Protein-only medium and hard complexes in the common subset predicted by AlphaFold 3, MultiFOLD, and SWISS-MODEL.",
        "n": 392,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "cameo-2024-ppi-three-server-common",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Figure 2 distributions are not converted into unsupported means or ranks.",
        "status": "verified"
      },
      "work_id": "cameo-paper",
      "work_version_id": "cameo-paper-2025-09-28"
    },
    {
      "benchmark_id": "casp-protein-ligands",
      "benchmark_version": "CASP16",
      "comparability_group": "casp16-ligand-affinity-stage1",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "casp16-ligand-affinity-stage1-evidence",
          "locator": {
            "note": "Original 140 cases, 18 disclosure exclusions, Stage-1 setting, Model 1, and N-weighted Kendall tau.",
            "type": "section",
            "value": "Sections 2.2–2.5 and 3.3; Figure 8A"
          },
          "source_id": "casp16-ligand-assessment",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        }
      ],
      "id": "casp16-ligand-affinity-stage1",
      "metrics": [
        {
          "aggregation": "case-count-weighted mean of per-protein-system Kendall tau after disclosure filtering",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "n-weighted-kendall-tau",
          "pass_threshold": null,
          "range": [
            -1,
            1
          ],
          "source_label": "N-weighted Kendall's tau",
          "tolerance": null,
          "unit": "rank correlation"
        }
      ],
      "model_ids": [],
      "protocol": {
        "contamination": {
          "notes": "Explicit patent-disclosure filtering is part of the final analysis set.",
          "reporting_status": "reported",
          "value": "experimental poses and affinities withheld, with 18 previously disclosed affinities removed from analysis"
        },
        "grader": {
          "human_review": true,
          "model": null,
          "reporting_status": "reported",
          "type": "official affinity ranking analysis plus independent ligand assessor team"
        },
        "reasoning": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "repeats": {
          "notes": "Not a repeated-episode benchmark.",
          "reporting_status": "reported",
          "value": "Model 1 affinity submission per group and case"
        },
        "seed": {
          "notes": "Method-specific stochastic settings are not normalized.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "statistical": {
          "notes": "Per-system Kendall tau is weighted by the number of retained affinity cases.",
          "reporting_status": "reported",
          "value": "N-weighted Kendall tau across the two protein systems after disclosure filtering"
        },
        "system_prompt_public": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "temperature": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "time_budget": {
          "notes": "No common compute/cost budget.",
          "reporting_status": "reported",
          "value": "target-specific CASP submission windows before experimental poses were released"
        },
        "token_budget": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "No common browser constraint.",
            "reporting_status": "reported",
            "value": "method-specific"
          },
          "code_execution": {
            "notes": "Affinity prediction pipelines execute participant code.",
            "reporting_status": "reported",
            "value": "allowed"
          },
          "container": {
            "notes": "No common container is specified.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Participant methods may use public chemical and structural data.",
            "reporting_status": "reported",
            "value": "method-specific"
          },
          "external_tools": {
            "notes": "Participants could report absolute affinity, relative affinity, or within-system ligand ranks.",
            "reporting_status": "reported",
            "value": "method-specific"
          },
          "internet": {
            "notes": "No common internet constraint.",
            "reporting_status": "reported",
            "value": "method-specific"
          }
        },
        "turns": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        }
      },
      "results": [],
      "scope": {
        "filter": "14 chymase and 108 autotaxin cases retained after removing 18 affinities disclosed in patents from the original 140.",
        "n": 122,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "casp16-affinity-stage1-analysis",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Stage 1 is isolated from Stage 2 because experimental co-crystal structures were unavailable in this setting.",
        "status": "verified"
      },
      "work_id": "casp16-ligand-assessment",
      "work_version_id": "casp16-ligand-assessment-2025-10-04"
    },
    {
      "benchmark_id": "casp-protein-ligands",
      "benchmark_version": "CASP16",
      "comparability_group": "casp16-ligand-affinity-stage2",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "casp16-ligand-affinity-stage2-evidence",
          "locator": {
            "note": "110 releases, 103 retained cases, co-crystal structure access, Stage-2 dates, and Kendall tau analysis.",
            "type": "section",
            "value": "Sections 2.2–2.5 and 3.3; Figures 8B-C and 9"
          },
          "source_id": "casp16-ligand-assessment",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        }
      ],
      "id": "casp16-ligand-affinity-stage2",
      "metrics": [
        {
          "aggregation": "case-count-weighted mean of per-protein-system Kendall tau after disclosure filtering",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "n-weighted-kendall-tau",
          "pass_threshold": null,
          "range": [
            -1,
            1
          ],
          "source_label": "N-weighted Kendall's tau",
          "tolerance": null,
          "unit": "rank correlation"
        }
      ],
      "model_ids": [],
      "protocol": {
        "contamination": {
          "notes": "103 undisclosed cases remain from 110 Stage-2 releases.",
          "reporting_status": "reported",
          "value": "experimental co-crystal structures provided while experimental affinities remained withheld; disclosed affinities removed"
        },
        "grader": {
          "human_review": true,
          "model": null,
          "reporting_status": "reported",
          "type": "official affinity ranking analysis plus independent ligand assessor team"
        },
        "reasoning": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "repeats": {
          "notes": "Not a repeated-episode benchmark.",
          "reporting_status": "reported",
          "value": "Model 1 affinity submission per group and case"
        },
        "seed": {
          "notes": "Method-specific stochastic settings are not normalized.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "statistical": {
          "notes": "Direct comparison with Stage 1 is reported, but the tool/input setting is different and therefore has a separate comparability group.",
          "reporting_status": "reported",
          "value": "per-system Kendall tau and N-weighted cross-system summary after disclosure filtering"
        },
        "system_prompt_public": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "temperature": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "time_budget": {
          "notes": "The late WDR55 pose target is not part of this affinity analysis.",
          "reporting_status": "reported",
          "value": "2024-08-07 through 2024-08-21 for the main Stage-2 challenge"
        },
        "token_budget": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "No common browser constraint.",
            "reporting_status": "reported",
            "value": "method-specific"
          },
          "code_execution": {
            "notes": "Affinity scoring pipelines execute participant code.",
            "reporting_status": "reported",
            "value": "allowed"
          },
          "container": {
            "notes": "No common container is specified.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Participant methods may use public chemical and structural data.",
            "reporting_status": "reported",
            "value": "method-specific"
          },
          "external_tools": {
            "notes": "Experimental protein-ligand co-crystal structures were supplied for the Stage-2 cases.",
            "reporting_status": "reported",
            "value": "method-specific"
          },
          "internet": {
            "notes": "No common internet constraint.",
            "reporting_status": "reported",
            "value": "method-specific"
          }
        },
        "turns": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        }
      },
      "results": [],
      "scope": {
        "filter": "14 chymase plus 89 autotaxin cases with experimental co-crystal structures and undisclosed affinities; 110 cases were released before disclosure filtering.",
        "n": 103,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "casp16-affinity-stage2-analysis",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Stage 2 is separate from Stage 1 because the experimental co-crystal structures were provided.",
        "status": "verified"
      },
      "work_id": "casp16-ligand-assessment",
      "work_version_id": "casp16-ligand-assessment-2025-10-04"
    },
    {
      "benchmark_id": "casp-protein-ligands",
      "benchmark_version": "CASP16",
      "comparability_group": "casp16-ligand-pose-regular",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "casp16-ligand-pose-run-evidence",
          "locator": {
            "note": "229 assessed pose cases, participant protocol, Model 1 handling, pose metrics, missing submissions, and separate baselines.",
            "type": "section",
            "value": "Abstract; Sections 2.1–2.5 and 3.1–3.2"
          },
          "source_id": "casp16-ligand-assessment",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        }
      ],
      "id": "casp16-ligand-pose-regular",
      "metrics": [
        {
          "aggregation": "mean across eligible pose targets with missing-prediction handling",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "lddt-pli",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "LDDT-PLI",
          "tolerance": "CASP16 modified the CASP15 definition to penalize erroneous non-native contacts.",
          "unit": "proportion"
        },
        {
          "aggregation": "per pose then group summary",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "bisyrmsd",
          "pass_threshold": null,
          "range": null,
          "source_label": "BiSyRMSD",
          "tolerance": "symmetry-corrected after binding-site superposition",
          "unit": "angstrom"
        },
        {
          "aggregation": "per binding pocket then group summary",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "lddt-lp",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "LDDT-lp",
          "tolerance": null,
          "unit": "proportion"
        },
        {
          "aggregation": "binding-site C-alpha RMSD then group summary",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "bb-rmsd",
          "pass_threshold": null,
          "range": null,
          "source_label": "BB-RMSD",
          "tolerance": null,
          "unit": "angstrom"
        }
      ],
      "model_ids": [],
      "protocol": {
        "contamination": {
          "notes": "Baseline predictions such as AlphaFold 3 are kept distinct from official participant submissions.",
          "reporting_status": "reported",
          "value": "experimental complex poses withheld during Stage 1 prediction; structural template similarity analyzed post hoc"
        },
        "grader": {
          "human_review": true,
          "model": null,
          "reporting_status": "reported",
          "type": "official OpenStructure ligand and binding-pocket scoring plus independent ligand assessor team"
        },
        "reasoning": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "repeats": {
          "notes": "Assessment emphasizes Model 1 while also inspecting best submitted poses.",
          "reporting_status": "reported",
          "value": "up to five pose models per target"
        },
        "seed": {
          "notes": "Method-specific stochastic settings are not normalized.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "statistical": {
          "notes": "The paper analyzes both participant groups and separate post hoc baselines.",
          "reporting_status": "reported",
          "value": "mean per-group pose accuracy with explicit handling of skipped targets; Model 1 is the primary ranking view"
        },
        "system_prompt_public": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "temperature": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "time_budget": {
          "notes": "The assessment does not normalize a compute or cost budget across groups.",
          "reporting_status": "reported",
          "value": "target-specific CASP submission windows"
        },
        "token_budget": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Participant methods are unrestricted beyond the blind-target protocol.",
            "reporting_status": "reported",
            "value": "method-specific"
          },
          "code_execution": {
            "notes": "Docking/cofolding pipelines execute participant code.",
            "reporting_status": "reported",
            "value": "allowed"
          },
          "container": {
            "notes": "No common container is specified.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Template and structure databases may be used; assessors analyze template similarity.",
            "reporting_status": "reported",
            "value": "method-specific"
          },
          "external_tools": {
            "notes": "Template-based, docking, cofolding, and hybrid workflows were submitted.",
            "reporting_status": "reported",
            "value": "method-specific"
          },
          "internet": {
            "notes": "No common internet restriction is reported.",
            "reporting_status": "reported",
            "value": "method-specific"
          }
        },
        "turns": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        }
      },
      "results": [],
      "scope": {
        "filter": "Final pharmaceutical pose assessment set; four of 233 released pose cases were not part of the paper assessment.",
        "n": 229,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "casp16-pharma-pose-assessed",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Released (233) and assessed (229) counts remain distinct; participant rows and post hoc baseline rows are not normalized in this PR.",
        "status": "verified"
      },
      "work_id": "casp16-ligand-assessment",
      "work_version_id": "casp16-ligand-assessment-2025-10-04"
    },
    {
      "benchmark_id": "casp-protein-monomers",
      "benchmark_version": "CASP16",
      "comparability_group": "casp16-monomer-regular-official",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "casp16-monomer-run-evidence-scope",
          "locator": {
            "note": "54 evaluation units, model-1/best-model analyses, metrics, and assessor interpretation.",
            "type": "section",
            "value": "Results and Discussion; performance evaluation and ranking"
          },
          "source_id": "casp16-monomer-assessment",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/scope",
            "/protocol/grader",
            "/protocol/statistical",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "casp16-monomer-run-evidence-protocol",
          "locator": {
            "note": "Blind targets, regular versus server settings, deadlines, and independent assessment.",
            "type": "section",
            "value": "Registration; Targets; Model submission/format; Assessment"
          },
          "source_id": "casp-monomer-casp16-home-resource",
          "source_type": "resource",
          "supports": [
            "/protocol/shots",
            "/protocol/turns",
            "/protocol/system_prompt_public",
            "/protocol/reasoning",
            "/protocol/tools",
            "/protocol/token_budget",
            "/protocol/time_budget",
            "/protocol/temperature",
            "/protocol/seed",
            "/protocol/repeats",
            "/protocol/contamination"
          ]
        }
      ],
      "id": "casp16-monomer-regular-official",
      "metrics": [
        {
          "aggregation": "computed per evaluation unit; official group rankings use first/best model views and z-score summaries",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "gdt-ts",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "GDT_TS",
          "tolerance": null,
          "unit": "score"
        },
        {
          "aggregation": "per evaluation unit",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "gdt-ha",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "GDT_HA",
          "tolerance": null,
          "unit": "score"
        },
        {
          "aggregation": "per evaluation unit",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "lddt",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "lDDT",
          "tolerance": null,
          "unit": "proportion"
        },
        {
          "aggregation": "per evaluation unit",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "tm-score",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "TM-score",
          "tolerance": null,
          "unit": "proportion"
        },
        {
          "aggregation": "sum across eligible evaluation units after clipping values below -2.0",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "cumulative-z-score",
          "pass_threshold": null,
          "range": null,
          "source_label": "SUM Z-score (> -2.0)",
          "tolerance": null,
          "unit": "z-score sum"
        }
      ],
      "model_ids": [],
      "protocol": {
        "contamination": {
          "notes": "Blind target handling is the core decontamination mechanism; training-set similarity is analyzed separately by assessors.",
          "reporting_status": "reported",
          "value": "experimental structures withheld until prediction closes"
        },
        "grader": {
          "human_review": true,
          "model": null,
          "reporting_status": "reported",
          "type": "official CASP structure-comparison pipeline plus independent monomer assessor team"
        },
        "reasoning": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "repeats": {
          "notes": "These are ranked model submissions, not repeated stochastic episodes; assessment reports first-model and best-model views.",
          "reporting_status": "reported",
          "value": "up to five submitted models per target"
        },
        "seed": {
          "notes": "Method-specific stochastic settings are not normalized.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "statistical": {
          "notes": "The official site exposes both raw per-model measures and group-level z-score rankings.",
          "reporting_status": "reported",
          "value": "per-evaluation-unit scores converted to group ranking summaries including cumulative clipped z-scores"
        },
        "system_prompt_public": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "temperature": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "time_budget": {
          "notes": "Server predictions use a separate approximately 72-hour setting and are not silently pooled into the regular-group ranking.",
          "reporting_status": "reported",
          "value": "approximately three weeks per target for regular groups"
        },
        "token_budget": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Regular groups may combine human knowledge and computational methods; no common browser setting is imposed.",
            "reporting_status": "reported",
            "value": "method-specific"
          },
          "code_execution": {
            "notes": "Computational structure-prediction pipelines are the evaluated methods.",
            "reporting_status": "reported",
            "value": "allowed"
          },
          "container": {
            "notes": "No common container is specified.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Methods may use public sequence and structure databases subject to target blindness.",
            "reporting_status": "reported",
            "value": "method-specific"
          },
          "external_tools": {
            "notes": "Each participant documents its own modeling pipeline.",
            "reporting_status": "reported",
            "value": "method-specific"
          },
          "internet": {
            "notes": "No common internet restriction is reported for regular groups.",
            "reporting_status": "reported",
            "value": "method-specific"
          }
        },
        "turns": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        }
      },
      "results": [],
      "scope": {
        "filter": "All 54 monomer evaluation units in the final CASP16 assessor analysis; regular-group rankings are distinguished from server-only rankings.",
        "n": 54,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Scope and metrics are normalized; individual group result rows are intentionally not ingested in this audit PR.",
        "status": "verified"
      },
      "work_id": "casp16-monomer-assessment",
      "work_version_id": "casp16-monomer-assessment-2025-08-17"
    },
    {
      "benchmark_id": "casp-protein-multimers",
      "benchmark_version": "CASP16",
      "comparability_group": "casp16-multimer-phase1-regular",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "casp16-multimer-run-evidence-scope",
          "locator": {
            "note": "40 Phase-1 targets, supplied stoichiometry, overall/interface metrics, first/best model analyses, and group ranking aggregation.",
            "type": "section",
            "value": "Overview of targets; Phase 1 Ranking; Performance Evaluation and Ranking"
          },
          "source_id": "casp16-multimer-assessment",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/scope",
            "/protocol/grader",
            "/protocol/statistical",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "casp16-multimer-run-evidence-protocol",
          "locator": {
            "note": "Blind regular-group protocol, participant method freedom, deadlines, and assessors.",
            "type": "section",
            "value": "Protein Complexes; Registration; Model submission/format; Assessment"
          },
          "source_id": "casp-multimer-casp16-home-resource",
          "source_type": "resource",
          "supports": [
            "/protocol/shots",
            "/protocol/turns",
            "/protocol/system_prompt_public",
            "/protocol/reasoning",
            "/protocol/tools",
            "/protocol/token_budget",
            "/protocol/time_budget",
            "/protocol/temperature",
            "/protocol/seed",
            "/protocol/repeats",
            "/protocol/contamination"
          ]
        }
      ],
      "id": "casp16-multimer-phase1-regular",
      "metrics": [
        {
          "aggregation": "per target followed by cumulative z-score ranking",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "dockq",
          "pass_threshold": 0.5,
          "range": [
            0,
            1
          ],
          "source_label": "DockQ",
          "tolerance": "The assessor paper uses DockQ >= 0.5 as a medium-or-better complex model threshold in summary analyses.",
          "unit": "proportion"
        },
        {
          "aggregation": "per target followed by cumulative z-score ranking",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "tm-score",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "TM-score",
          "tolerance": null,
          "unit": "proportion"
        },
        {
          "aggregation": "per target followed by cumulative z-score ranking",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "lddt",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "lDDT",
          "tolerance": null,
          "unit": "proportion"
        },
        {
          "aggregation": "per target followed by cumulative z-score ranking",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "interface-contact-score",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "ICS",
          "tolerance": null,
          "unit": "F1 score"
        },
        {
          "aggregation": "per target followed by cumulative z-score ranking",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "interface-patch-score",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "IPS",
          "tolerance": null,
          "unit": "Jaccard score"
        },
        {
          "aggregation": "per target followed by cumulative z-score ranking",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "qs-best",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "QSbest",
          "tolerance": null,
          "unit": "Jaccard score"
        }
      ],
      "model_ids": [],
      "protocol": {
        "contamination": {
          "notes": "Template availability and homologous information are analyzed by assessors rather than treated as identical across targets.",
          "reporting_status": "reported",
          "value": "experimental complex structures withheld until prediction closes"
        },
        "grader": {
          "human_review": true,
          "model": null,
          "reporting_status": "reported",
          "type": "official CASP/OpenStructure and CAPRI-compatible scoring plus independent complex assessor team"
        },
        "reasoning": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "repeats": {
          "notes": "Official analyses distinguish first submitted and best submitted models.",
          "reporting_status": "reported",
          "value": "up to five standard models per target"
        },
        "seed": {
          "notes": "Method-specific stochastic settings are not normalized.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "statistical": {
          "notes": "Metrics capture both overall folds and interfaces.",
          "reporting_status": "reported",
          "value": "cumulative per-target z-scores and head-to-head comparisons; rankings reported for first and best models"
        },
        "system_prompt_public": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "temperature": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "time_budget": {
          "notes": "Phase-1 regular predictions are separated from short-deadline server settings.",
          "reporting_status": "reported",
          "value": "approximately three weeks per target for regular groups"
        },
        "token_budget": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Regular groups may combine human and computational workflows.",
            "reporting_status": "reported",
            "value": "method-specific"
          },
          "code_execution": {
            "notes": "Participant prediction pipelines execute code.",
            "reporting_status": "reported",
            "value": "allowed"
          },
          "container": {
            "notes": "No common container is specified.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Public sequence/structure databases may be used while target structures remain blind.",
            "reporting_status": "reported",
            "value": "method-specific"
          },
          "external_tools": {
            "notes": "Participant methods commonly use AFM/AF3-derived pipelines, enhanced MSAs, sampling, and model selection.",
            "reporting_status": "reported",
            "value": "method-specific"
          },
          "internet": {
            "notes": "No common internet restriction is imposed.",
            "reporting_status": "reported",
            "value": "method-specific"
          }
        },
        "turns": {
          "notes": "No natural-language in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        }
      },
      "results": [],
      "scope": {
        "filter": "All 40 unique protein-complex targets in the Phase-1 main assessment, where experimental stoichiometry was provided.",
        "n": 40,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Phase 1 is isolated from the stoichiometry-free Phase 0 and MassiveFold Phase 2 settings; individual group rows are not ingested here.",
        "status": "verified"
      },
      "work_id": "casp16-multimer-assessment",
      "work_version_id": "casp16-multimer-assessment-2025-10-31"
    },
    {
      "benchmark_id": "compbiobench",
      "benchmark_version": "v1",
      "comparability_group": "compbiobench-v1-hardest-codex-xhigh-three-runs",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "compbiobench-evidence-codex-hardest-protocol",
          "locator": {
            "note": "Exact Codex configuration and common agent protocol.",
            "type": "page",
            "value": "PDF pp. 13–14, Methods"
          },
          "source_id": "compbiobench-preprint",
          "source_type": "work",
          "supports": [
            "/protocol"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "compbiobench-evidence-codex-hardest-results",
          "locator": {
            "note": "Levels 4 and 5 grouped; 17 tasks; Codex label 59; three-run average.",
            "type": "figure",
            "value": "PDF p. 11, Supplementary Figure 2B and caption"
          },
          "source_id": "compbiobench-preprint",
          "source_type": "work",
          "supports": [
            "/scope",
            "/metrics",
            "/results"
          ]
        }
      ],
      "id": "compbiobench-codex-hardest",
      "metrics": [
        {
          "aggregation": "mean problem accuracy over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Accuracy — difficulty Levels 4–5",
          "tolerance": "whitespace-stripped exact match",
          "unit": "percent"
        }
      ],
      "model_ids": [
        "codex-cli-gpt-5-4"
      ],
      "protocol": {
        "contamination": {
          "notes": "No pretraining decontamination analysis.",
          "reporting_status": "reported",
          "value": "benchmark construction reduces lookup shortcuts"
        },
        "grader": {
          "human_review": true,
          "model": null,
          "reporting_status": "reported",
          "type": "whitespace-stripped exact string match with occasional manual format-only correction"
        },
        "reasoning": {
          "notes": "gpt-5.4 model_reasoning_effort=xhigh.",
          "reporting_status": "reported",
          "value": "xhigh"
        },
        "repeats": {
          "notes": "Values averaged across three runs.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public wrapper prompt has no examples.",
          "reporting_status": "reported",
          "value": "zero-shot"
        },
        "statistical": {
          "notes": "No CI.",
          "reporting_status": "reported",
          "value": "accuracy averaged over three runs within the 17-task subset"
        },
        "system_prompt_public": {
          "notes": "Benchmark wrapper public; vendor system prompt not public.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Same underlying runs as the full result.",
          "reporting_status": "reported",
          "value": "120 minutes initially; timeout rerun once with 240 minutes"
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "No distinct browser UI specified.",
            "reporting_status": "reported",
            "value": "web access through agent environment"
          },
          "code_execution": {
            "notes": "Auto-approved command execution.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "Per-question Conda clone, not a hard container.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "External biological resources permitted.",
            "reporting_status": "reported",
            "value": true
          },
          "external_tools": {
            "notes": "Installation/retrieval allowed.",
            "reporting_status": "reported",
            "value": true
          },
          "internet": {
            "notes": "Web enabled.",
            "reporting_status": "reported",
            "value": true
          }
        },
        "turns": {
          "notes": "Codex agent.",
          "reporting_status": "reported",
          "value": "multi-turn agent trajectory"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "compbiobench-evidence-codex-hardest-results"
          ],
          "metric_id": "accuracy",
          "model_id": "codex-cli-gpt-5-4",
          "n": 17,
          "notes": "Supplementary Figure 2 label; average across three runs.",
          "status": "verified",
          "value": 59.0
        }
      ],
      "scope": {
        "filter": "Contributor-rated difficulty Levels 4 and 5 grouped together.",
        "n": 17,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "compbiobench-difficulty-levels-4-5",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Seventeen-task grouped scope and 59% three-run mean verified.",
        "status": "verified"
      },
      "work_id": "compbiobench-preprint",
      "work_version_id": "compbiobench-preprint-2026-04-09"
    },
    {
      "benchmark_id": "compbiobench",
      "benchmark_version": "v1",
      "comparability_group": "compbiobench-v1-codex-xhigh-three-runs",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "compbiobench-evidence-codex-scope-protocol",
          "locator": {
            "note": "Reports public wrapper prompt, internet/code/tool access, Conda isolation, 120/240-minute policy, Codex CLI v0.115.0, gpt-5.4, xhigh, and three runs.",
            "type": "page",
            "value": "PDF pp. 13–14, Methods: agent execution and model configurations"
          },
          "source_id": "compbiobench-preprint",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/scope",
            "/protocol"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "compbiobench-evidence-codex-results",
          "locator": {
            "note": "Printed labels report 83.3% accuracy, 679.0 seconds, and USD 1.0; text reports three-run consistency.",
            "type": "figure",
            "value": "PDF pp. 3–4, Figure 2A–C and Results"
          },
          "source_id": "compbiobench-preprint",
          "source_type": "work",
          "supports": [
            "/metrics",
            "/results"
          ]
        }
      ],
      "id": "compbiobench-creator-full",
      "metrics": [
        {
          "aggregation": "mean problem accuracy over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Accuracy",
          "tolerance": "whitespace-stripped exact match",
          "unit": "percent"
        },
        {
          "aggregation": "mean over questions and runs after 7200-second display clipping",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "mean-wall-clock-time",
          "pass_threshold": null,
          "range": null,
          "source_label": "Wall-clock time per question",
          "tolerance": null,
          "unit": "seconds"
        },
        {
          "aggregation": "mean over questions and runs after USD-10 display clipping",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "mean-cost",
          "pass_threshold": null,
          "range": null,
          "source_label": "Cost per question",
          "tolerance": null,
          "unit": "USD"
        }
      ],
      "model_ids": [
        "codex-cli-gpt-5-4"
      ],
      "protocol": {
        "contamination": {
          "notes": "No model pretraining decontamination analysis is reported.",
          "reporting_status": "reported",
          "value": "synthetic/augmented inputs and scrubbed identifiers reduce lookup shortcuts"
        },
        "grader": {
          "human_review": true,
          "model": null,
          "reporting_status": "reported",
          "type": "whitespace-stripped exact string match with occasional manual format-only correction"
        },
        "reasoning": {
          "notes": "model_reasoning_effort=xhigh.",
          "reporting_status": "reported",
          "value": "xhigh"
        },
        "repeats": {
          "notes": "Three independent full-benchmark runs.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published wrapper prompt contains instructions but no worked examples.",
          "reporting_status": "reported",
          "value": "zero-shot"
        },
        "statistical": {
          "notes": "Wall time and cost plots clip per-question values at 7200 seconds and USD 10; no confidence interval is published.",
          "reporting_status": "reported",
          "value": "mean over three independent runs; consistency reported as correct in all three and correct at least once"
        },
        "system_prompt_public": {
          "notes": "The benchmark wrapper prompt is public; the underlying vendor system prompt is not published.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "A second timeout was scored incorrect.",
          "reporting_status": "reported",
          "value": "120 minutes initially; timed-out questions rerun clean once with 240 minutes"
        },
        "token_budget": {
          "notes": "No token ceiling is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "No distinct browser UI is specified.",
            "reporting_status": "reported",
            "value": "web access through the agent environment"
          },
          "code_execution": {
            "notes": "Codex ran arbitrary analysis commands with auto approval.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "A per-question Conda environment was cloned; the runner was not a hard filesystem sandbox or container.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "Agents could retrieve external biological resources.",
            "reporting_status": "reported",
            "value": true
          },
          "external_tools": {
            "notes": "Agents could install and retrieve required tools.",
            "reporting_status": "reported",
            "value": true
          },
          "internet": {
            "notes": "Codex network_access was enabled and web use was encouraged.",
            "reporting_status": "reported",
            "value": true
          }
        },
        "turns": {
          "notes": "Codex autonomously used the shell, code, tools, and web before returning one answer.",
          "reporting_status": "reported",
          "value": "multi-turn agent trajectory"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "compbiobench-evidence-codex-results"
          ],
          "metric_id": "accuracy",
          "model_id": "codex-cli-gpt-5-4",
          "n": 100,
          "notes": "Mean of three full runs; 73% solved on all three and 92% at least once.",
          "status": "verified",
          "value": 83.3
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "compbiobench-evidence-codex-results"
          ],
          "metric_id": "mean-wall-clock-time",
          "model_id": "codex-cli-gpt-5-4",
          "n": 100,
          "notes": "Printed Figure 2 label; mean across three runs.",
          "status": "verified",
          "value": 679.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "compbiobench-evidence-codex-results"
          ],
          "metric_id": "mean-cost",
          "model_id": "codex-cli-gpt-5-4",
          "n": 100,
          "notes": "Printed Figure 2 label; mean across three runs.",
          "status": "verified",
          "value": 1.0
        }
      ],
      "scope": {
        "filter": "All 100 v1 tasks.",
        "n": 100,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Full scope, exact CLI/model/effort, repeat protocol, and three Figure 2 values verified.",
        "status": "verified"
      },
      "work_id": "compbiobench-preprint",
      "work_version_id": "compbiobench-preprint-2026-04-09"
    },
    {
      "benchmark_id": "compbiobench",
      "benchmark_version": "v1",
      "comparability_group": "compbiobench-v1-haiku-one-run-120m",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "compbiobench-evidence-haiku-scope-protocol",
          "locator": {
            "note": "Reports Claude Code v2.1.87, dated Haiku identifier, unsupported effort, one run, and no timeout rerun.",
            "type": "page",
            "value": "PDF pp. 13–14, Methods"
          },
          "source_id": "compbiobench-preprint",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/scope",
            "/protocol"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "compbiobench-evidence-haiku-results",
          "locator": {
            "note": "Printed labels report 34.0%, 809.3 seconds, and USD 0.3.",
            "type": "figure",
            "value": "PDF pp. 3–4, Figure 2A–C"
          },
          "source_id": "compbiobench-preprint",
          "source_type": "work",
          "supports": [
            "/metrics",
            "/results"
          ]
        }
      ],
      "id": "compbiobench-haiku-full",
      "metrics": [
        {
          "aggregation": "problem-weighted mean",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Accuracy",
          "tolerance": "whitespace-stripped exact match",
          "unit": "percent"
        },
        {
          "aggregation": "mean after 7200-second display clipping",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "mean-wall-clock-time",
          "pass_threshold": null,
          "range": null,
          "source_label": "Wall-clock time per question",
          "tolerance": null,
          "unit": "seconds"
        },
        {
          "aggregation": "mean after USD-10 display clipping",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "mean-cost",
          "pass_threshold": null,
          "range": null,
          "source_label": "Cost per question",
          "tolerance": null,
          "unit": "USD"
        }
      ],
      "model_ids": [
        "claude-code-haiku-4-5"
      ],
      "protocol": {
        "contamination": {
          "notes": "No pretraining decontamination analysis.",
          "reporting_status": "reported",
          "value": "synthetic/augmented inputs and scrubbed identifiers reduce lookup shortcuts"
        },
        "grader": {
          "human_review": true,
          "model": null,
          "reporting_status": "reported",
          "type": "whitespace-stripped exact string match with occasional manual format-only correction"
        },
        "reasoning": {
          "notes": "The --effort setting was not supported for Haiku.",
          "reporting_status": "reported",
          "value": "unsupported"
        },
        "repeats": {
          "notes": "One full-benchmark run.",
          "reporting_status": "reported",
          "value": 1
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public wrapper prompt has no worked examples.",
          "reporting_status": "reported",
          "value": "zero-shot"
        },
        "statistical": {
          "notes": "Wall time/cost display clipping at 7200 seconds/USD 10; no CI.",
          "reporting_status": "reported",
          "value": "single full-benchmark run"
        },
        "system_prompt_public": {
          "notes": "Benchmark wrapper is public; vendor system prompt is not.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Haiku was the explicit exception to timeout reruns.",
          "reporting_status": "reported",
          "value": "120 minutes; no 240-minute timeout rerun"
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "No distinct browser UI specified.",
            "reporting_status": "reported",
            "value": "web access through the agent environment"
          },
          "code_execution": {
            "notes": "Commands auto-approved.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "Per-question Conda clone, not a hard sandbox/container.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "External biological resources permitted.",
            "reporting_status": "reported",
            "value": true
          },
          "external_tools": {
            "notes": "Tool installation and retrieval permitted.",
            "reporting_status": "reported",
            "value": true
          },
          "internet": {
            "notes": "Web access enabled and encouraged.",
            "reporting_status": "reported",
            "value": true
          }
        },
        "turns": {
          "notes": "Tool-using Claude Code trajectory.",
          "reporting_status": "reported",
          "value": "multi-turn agent trajectory"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "compbiobench-evidence-haiku-results"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-code-haiku-4-5",
          "n": 100,
          "notes": "One full run.",
          "status": "verified",
          "value": 34.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "compbiobench-evidence-haiku-results"
          ],
          "metric_id": "mean-wall-clock-time",
          "model_id": "claude-code-haiku-4-5",
          "n": 100,
          "notes": "Printed Figure 2 label.",
          "status": "verified",
          "value": 809.3
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "compbiobench-evidence-haiku-results"
          ],
          "metric_id": "mean-cost",
          "model_id": "claude-code-haiku-4-5",
          "n": 100,
          "notes": "Printed Figure 2 label.",
          "status": "verified",
          "value": 0.3
        }
      ],
      "scope": {
        "filter": "All 100 v1 tasks.",
        "n": 100,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact one-run Haiku setting, timeout exception, and Figure 2 values verified.",
        "status": "verified"
      },
      "work_id": "compbiobench-preprint",
      "work_version_id": "compbiobench-preprint-2026-04-09"
    },
    {
      "benchmark_id": "compbiobench",
      "benchmark_version": "v1",
      "comparability_group": "compbiobench-v1-hardest-haiku-one-run-120m",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "compbiobench-evidence-haiku-hardest-protocol",
          "locator": {
            "note": "Exact Haiku configuration and timeout exception.",
            "type": "page",
            "value": "PDF pp. 13–14, Methods"
          },
          "source_id": "compbiobench-preprint",
          "source_type": "work",
          "supports": [
            "/protocol"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "compbiobench-evidence-haiku-hardest-results",
          "locator": {
            "note": "Levels 4 and 5 grouped; 17 tasks; Haiku label 12.",
            "type": "figure",
            "value": "PDF p. 11, Supplementary Figure 2B and caption"
          },
          "source_id": "compbiobench-preprint",
          "source_type": "work",
          "supports": [
            "/scope",
            "/metrics",
            "/results"
          ]
        }
      ],
      "id": "compbiobench-haiku-hardest",
      "metrics": [
        {
          "aggregation": "problem-weighted mean",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Accuracy — difficulty Levels 4–5",
          "tolerance": "whitespace-stripped exact match",
          "unit": "percent"
        }
      ],
      "model_ids": [
        "claude-code-haiku-4-5"
      ],
      "protocol": {
        "contamination": {
          "notes": "No pretraining decontamination analysis.",
          "reporting_status": "reported",
          "value": "benchmark construction reduces lookup shortcuts"
        },
        "grader": {
          "human_review": true,
          "model": null,
          "reporting_status": "reported",
          "type": "whitespace-stripped exact string match with occasional manual format-only correction"
        },
        "reasoning": {
          "notes": "Effort unsupported for Haiku.",
          "reporting_status": "reported",
          "value": "unsupported"
        },
        "repeats": {
          "notes": "One run.",
          "reporting_status": "reported",
          "value": 1
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public wrapper prompt has no examples.",
          "reporting_status": "reported",
          "value": "zero-shot"
        },
        "statistical": {
          "notes": "No CI.",
          "reporting_status": "reported",
          "value": "problem-weighted accuracy in one run over the 17-task subset"
        },
        "system_prompt_public": {
          "notes": "Benchmark wrapper public; vendor system prompt not public.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Haiku exception.",
          "reporting_status": "reported",
          "value": "120 minutes; no 240-minute timeout rerun"
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "No distinct browser UI specified.",
            "reporting_status": "reported",
            "value": "web access through agent environment"
          },
          "code_execution": {
            "notes": "Auto-approved commands.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "Per-question Conda clone, not a hard container.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "External biological resources permitted.",
            "reporting_status": "reported",
            "value": true
          },
          "external_tools": {
            "notes": "Installation/retrieval allowed.",
            "reporting_status": "reported",
            "value": true
          },
          "internet": {
            "notes": "Web enabled.",
            "reporting_status": "reported",
            "value": true
          }
        },
        "turns": {
          "notes": "Claude Code agent.",
          "reporting_status": "reported",
          "value": "multi-turn agent trajectory"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "compbiobench-evidence-haiku-hardest-results"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-code-haiku-4-5",
          "n": 17,
          "notes": "Supplementary Figure 2 label; one run.",
          "status": "verified",
          "value": 12.0
        }
      ],
      "scope": {
        "filter": "Contributor-rated difficulty Levels 4 and 5 grouped together.",
        "n": 17,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "compbiobench-difficulty-levels-4-5",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Seventeen-task grouped scope and 12% one-run value verified.",
        "status": "verified"
      },
      "work_id": "compbiobench-preprint",
      "work_version_id": "compbiobench-preprint-2026-04-09"
    },
    {
      "benchmark_id": "compbiobench",
      "benchmark_version": "v1",
      "comparability_group": "compbiobench-v1-nonagentic-api-three-calls-no-files",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "compbiobench-evidence-nonagentic-protocol",
          "locator": {
            "note": "Default API parameters, three calls per question, no files, public prompt, and full benchmark.",
            "type": "page",
            "value": "PDF p. 13, Methods: LLM-only baselines"
          },
          "source_id": "compbiobench-preprint",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/scope",
            "/protocol"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "compbiobench-evidence-nonagentic-results",
          "locator": {
            "note": "Figure labels and text report ChatGPT 5.2 at 5.3% and Claude Opus 4.6 at 3.7%.",
            "type": "figure",
            "value": "PDF pp. 3–4, Figure 2A and Results"
          },
          "source_id": "compbiobench-preprint",
          "source_type": "work",
          "supports": [
            "/metrics",
            "/results"
          ]
        }
      ],
      "id": "compbiobench-nonagentic-baselines",
      "metrics": [
        {
          "aggregation": "mean over three calls per problem",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Accuracy",
          "tolerance": "whitespace-stripped exact match",
          "unit": "percent"
        }
      ],
      "model_ids": [
        "chatgpt-5-2",
        "claude-opus-4-6"
      ],
      "protocol": {
        "contamination": {
          "notes": "No pretraining decontamination analysis.",
          "reporting_status": "reported",
          "value": "benchmark construction reduces lookup shortcuts"
        },
        "grader": {
          "human_review": null,
          "model": null,
          "reporting_status": "reported",
          "type": "whitespace-stripped exact string match"
        },
        "reasoning": {
          "notes": "No explicit reasoning-effort setting reported.",
          "reporting_status": "reported",
          "value": "default API parameters"
        },
        "repeats": {
          "notes": "Every question was sent three times to each model.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Full public prompt with no demonstrations.",
          "reporting_status": "reported",
          "value": "zero-shot"
        },
        "statistical": {
          "notes": "No CI reported.",
          "reporting_status": "reported",
          "value": "mean accuracy over three calls per question"
        },
        "system_prompt_public": {
          "notes": "The creator prompt is public; provider system prompts are not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "temperature": {
          "notes": "Source says default API parameters but does not report a numeric temperature.",
          "reporting_status": "reported",
          "value": "default"
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "No browser.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "No code execution.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Direct API calls.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "No database access.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "No tools or input files.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "No web access.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "Non-agentic baselines did not run an iterative tool-using trajectory.",
          "reporting_status": "reported",
          "value": "single-turn API call"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "compbiobench-evidence-nonagentic-results"
          ],
          "metric_id": "accuracy",
          "model_id": "chatgpt-5-2",
          "n": 100,
          "notes": "ChatGPT 5.2 non-agentic API baseline; three calls per question.",
          "status": "verified",
          "value": 5.3
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "compbiobench-evidence-nonagentic-results"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-opus-4-6",
          "n": 100,
          "notes": "Claude Opus 4.6 non-agentic API baseline; three calls per question.",
          "status": "verified",
          "value": 3.7
        }
      ],
      "scope": {
        "filter": "All 100 v1 questions, presented without associated files.",
        "n": 100,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Both full-set baseline results and the no-files, no-tools, three-call API protocol verified.",
        "status": "verified"
      },
      "work_id": "compbiobench-preprint",
      "work_version_id": "compbiobench-preprint-2026-04-09"
    },
    {
      "benchmark_id": "compbiobench",
      "benchmark_version": "v1",
      "comparability_group": "compbiobench-v1-opus-max-three-runs",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "compbiobench-evidence-opus-scope-protocol",
          "locator": {
            "note": "Reports Claude Code v2.1.87, claude-opus-4-6 1M, max effort, three runs, tools, and timeout policy.",
            "type": "page",
            "value": "PDF pp. 13–14, Methods: agent execution and model configurations"
          },
          "source_id": "compbiobench-preprint",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/scope",
            "/protocol"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "compbiobench-evidence-opus-results",
          "locator": {
            "note": "Reports 81.0% accuracy, 1101.0 seconds, USD 1.7, and consistency.",
            "type": "figure",
            "value": "PDF pp. 3–4, Figure 2A–C and Results"
          },
          "source_id": "compbiobench-preprint",
          "source_type": "work",
          "supports": [
            "/metrics",
            "/results"
          ]
        }
      ],
      "id": "compbiobench-opus-full",
      "metrics": [
        {
          "aggregation": "mean problem accuracy over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Accuracy",
          "tolerance": "whitespace-stripped exact match",
          "unit": "percent"
        },
        {
          "aggregation": "mean after 7200-second display clipping",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "mean-wall-clock-time",
          "pass_threshold": null,
          "range": null,
          "source_label": "Wall-clock time per question",
          "tolerance": null,
          "unit": "seconds"
        },
        {
          "aggregation": "mean after USD-10 display clipping",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "mean-cost",
          "pass_threshold": null,
          "range": null,
          "source_label": "Cost per question",
          "tolerance": null,
          "unit": "USD"
        }
      ],
      "model_ids": [
        "claude-code-opus-4-6"
      ],
      "protocol": {
        "contamination": {
          "notes": "No pretraining decontamination analysis.",
          "reporting_status": "reported",
          "value": "synthetic/augmented inputs and scrubbed identifiers reduce lookup shortcuts"
        },
        "grader": {
          "human_review": true,
          "model": null,
          "reporting_status": "reported",
          "type": "whitespace-stripped exact string match with occasional manual format-only correction"
        },
        "reasoning": {
          "notes": "Claude Code --effort max.",
          "reporting_status": "reported",
          "value": "max"
        },
        "repeats": {
          "notes": "Three independent full-benchmark runs.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The published wrapper prompt contains no worked examples.",
          "reporting_status": "reported",
          "value": "zero-shot"
        },
        "statistical": {
          "notes": "Wall time/cost display clipping at 7200 seconds/USD 10; no CI published.",
          "reporting_status": "reported",
          "value": "mean over three independent runs; consistency reported as correct in all three and at least once"
        },
        "system_prompt_public": {
          "notes": "The benchmark wrapper prompt is public; the vendor system prompt is not.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "A second timeout was incorrect.",
          "reporting_status": "reported",
          "value": "120 minutes initially; timed-out questions rerun clean once with 240 minutes"
        },
        "token_budget": {
          "notes": "No token ceiling is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "No distinct browser UI is specified.",
            "reporting_status": "reported",
            "value": "web access through the agent environment"
          },
          "code_execution": {
            "notes": "Claude Code ran commands with --dangerously-skip-permissions.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "Per-question Conda clone, not a hard sandbox or container.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "External biological resources were permitted.",
            "reporting_status": "reported",
            "value": true
          },
          "external_tools": {
            "notes": "Agents could install and retrieve tools.",
            "reporting_status": "reported",
            "value": true
          },
          "internet": {
            "notes": "Web access was enabled and encouraged.",
            "reporting_status": "reported",
            "value": true
          }
        },
        "turns": {
          "notes": "Claude Code could use the shell, code, tools, and web before returning one answer.",
          "reporting_status": "reported",
          "value": "multi-turn agent trajectory"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "compbiobench-evidence-opus-results"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-code-opus-4-6",
          "n": 100,
          "notes": "Mean of three full runs; 73% solved all three and 86% at least once.",
          "status": "verified",
          "value": 81.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "compbiobench-evidence-opus-results"
          ],
          "metric_id": "mean-wall-clock-time",
          "model_id": "claude-code-opus-4-6",
          "n": 100,
          "notes": "Printed Figure 2 label.",
          "status": "verified",
          "value": 1101.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "compbiobench-evidence-opus-results"
          ],
          "metric_id": "mean-cost",
          "model_id": "claude-code-opus-4-6",
          "n": 100,
          "notes": "Printed Figure 2 label.",
          "status": "verified",
          "value": 1.7
        }
      ],
      "scope": {
        "filter": "All 100 v1 tasks.",
        "n": 100,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Full scope, exact Claude Code/model/effort, repeat protocol, and Figure 2 values verified.",
        "status": "verified"
      },
      "work_id": "compbiobench-preprint",
      "work_version_id": "compbiobench-preprint-2026-04-09"
    },
    {
      "benchmark_id": "compbiobench",
      "benchmark_version": "v1",
      "comparability_group": "compbiobench-v1-hardest-opus-max-three-runs",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "compbiobench-evidence-opus-hardest-protocol",
          "locator": {
            "note": "Exact Claude Code Opus configuration and common protocol.",
            "type": "page",
            "value": "PDF pp. 13–14, Methods"
          },
          "source_id": "compbiobench-preprint",
          "source_type": "work",
          "supports": [
            "/protocol"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "compbiobench-evidence-opus-hardest-results",
          "locator": {
            "note": "Levels 4 and 5 grouped; 17 tasks; Opus label 69; three-run average.",
            "type": "figure",
            "value": "PDF p. 11, Supplementary Figure 2B and caption"
          },
          "source_id": "compbiobench-preprint",
          "source_type": "work",
          "supports": [
            "/scope",
            "/metrics",
            "/results"
          ]
        }
      ],
      "id": "compbiobench-opus-hardest",
      "metrics": [
        {
          "aggregation": "mean problem accuracy over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Accuracy — difficulty Levels 4–5",
          "tolerance": "whitespace-stripped exact match",
          "unit": "percent"
        }
      ],
      "model_ids": [
        "claude-code-opus-4-6"
      ],
      "protocol": {
        "contamination": {
          "notes": "No pretraining decontamination analysis.",
          "reporting_status": "reported",
          "value": "benchmark construction reduces lookup shortcuts"
        },
        "grader": {
          "human_review": true,
          "model": null,
          "reporting_status": "reported",
          "type": "whitespace-stripped exact string match with occasional manual format-only correction"
        },
        "reasoning": {
          "notes": "Claude Code --effort max.",
          "reporting_status": "reported",
          "value": "max"
        },
        "repeats": {
          "notes": "Values averaged across three runs.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public wrapper prompt has no examples.",
          "reporting_status": "reported",
          "value": "zero-shot"
        },
        "statistical": {
          "notes": "No CI.",
          "reporting_status": "reported",
          "value": "accuracy averaged over three runs within the 17-task subset"
        },
        "system_prompt_public": {
          "notes": "Benchmark wrapper public; vendor system prompt not public.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Same runs as full result.",
          "reporting_status": "reported",
          "value": "120 minutes initially; timeout rerun once with 240 minutes"
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "No distinct browser UI specified.",
            "reporting_status": "reported",
            "value": "web access through agent environment"
          },
          "code_execution": {
            "notes": "Auto-approved commands.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "Per-question Conda clone, not a hard container.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "External biological resources permitted.",
            "reporting_status": "reported",
            "value": true
          },
          "external_tools": {
            "notes": "Installation/retrieval allowed.",
            "reporting_status": "reported",
            "value": true
          },
          "internet": {
            "notes": "Web enabled.",
            "reporting_status": "reported",
            "value": true
          }
        },
        "turns": {
          "notes": "Claude Code agent.",
          "reporting_status": "reported",
          "value": "multi-turn agent trajectory"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "compbiobench-evidence-opus-hardest-results"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-code-opus-4-6",
          "n": 17,
          "notes": "Supplementary Figure 2 label; average across three runs.",
          "status": "verified",
          "value": 69.0
        }
      ],
      "scope": {
        "filter": "Contributor-rated difficulty Levels 4 and 5 grouped together.",
        "n": 17,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "compbiobench-difficulty-levels-4-5",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Seventeen-task grouped scope and 69% three-run mean verified.",
        "status": "verified"
      },
      "work_id": "compbiobench-preprint",
      "work_version_id": "compbiobench-preprint-2026-04-09"
    },
    {
      "benchmark_id": "compbiobench",
      "benchmark_version": "v1",
      "comparability_group": "compbiobench-v1-sonnet-high-one-run",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "compbiobench-evidence-sonnet-scope-protocol",
          "locator": {
            "note": "Reports Claude Code v2.1.87, claude-sonnet-4-6 1M, high effort, one run, tools, and timeout policy.",
            "type": "page",
            "value": "PDF pp. 13–14, Methods"
          },
          "source_id": "compbiobench-preprint",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/scope",
            "/protocol"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "compbiobench-evidence-sonnet-results",
          "locator": {
            "note": "Printed labels report 70.0%, 1049.2 seconds, and USD 1.2.",
            "type": "figure",
            "value": "PDF pp. 3–4, Figure 2A–C"
          },
          "source_id": "compbiobench-preprint",
          "source_type": "work",
          "supports": [
            "/metrics",
            "/results"
          ]
        }
      ],
      "id": "compbiobench-sonnet-full",
      "metrics": [
        {
          "aggregation": "problem-weighted mean",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Accuracy",
          "tolerance": "whitespace-stripped exact match",
          "unit": "percent"
        },
        {
          "aggregation": "mean after 7200-second display clipping",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "mean-wall-clock-time",
          "pass_threshold": null,
          "range": null,
          "source_label": "Wall-clock time per question",
          "tolerance": null,
          "unit": "seconds"
        },
        {
          "aggregation": "mean after USD-10 display clipping",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "mean-cost",
          "pass_threshold": null,
          "range": null,
          "source_label": "Cost per question",
          "tolerance": null,
          "unit": "USD"
        }
      ],
      "model_ids": [
        "claude-code-sonnet-4-6"
      ],
      "protocol": {
        "contamination": {
          "notes": "No pretraining decontamination analysis.",
          "reporting_status": "reported",
          "value": "synthetic/augmented inputs and scrubbed identifiers reduce lookup shortcuts"
        },
        "grader": {
          "human_review": true,
          "model": null,
          "reporting_status": "reported",
          "type": "whitespace-stripped exact string match with occasional manual format-only correction"
        },
        "reasoning": {
          "notes": "Claude Code --effort high.",
          "reporting_status": "reported",
          "value": "high"
        },
        "repeats": {
          "notes": "One full-benchmark run.",
          "reporting_status": "reported",
          "value": 1
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public wrapper prompt has no worked examples.",
          "reporting_status": "reported",
          "value": "zero-shot"
        },
        "statistical": {
          "notes": "Wall time/cost display clipping at 7200 seconds/USD 10; no CI.",
          "reporting_status": "reported",
          "value": "single full-benchmark run"
        },
        "system_prompt_public": {
          "notes": "Benchmark wrapper is public; vendor system prompt is not.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "A second timeout was incorrect.",
          "reporting_status": "reported",
          "value": "120 minutes initially; timed-out questions rerun clean once with 240 minutes"
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "No distinct browser UI specified.",
            "reporting_status": "reported",
            "value": "web access through the agent environment"
          },
          "code_execution": {
            "notes": "Commands auto-approved.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "Per-question Conda clone, not a hard sandbox/container.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "External biological resources permitted.",
            "reporting_status": "reported",
            "value": true
          },
          "external_tools": {
            "notes": "Tool installation and retrieval permitted.",
            "reporting_status": "reported",
            "value": true
          },
          "internet": {
            "notes": "Web access enabled and encouraged.",
            "reporting_status": "reported",
            "value": true
          }
        },
        "turns": {
          "notes": "Tool-using Claude Code trajectory.",
          "reporting_status": "reported",
          "value": "multi-turn agent trajectory"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "compbiobench-evidence-sonnet-results"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-code-sonnet-4-6",
          "n": 100,
          "notes": "One full run.",
          "status": "verified",
          "value": 70.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "compbiobench-evidence-sonnet-results"
          ],
          "metric_id": "mean-wall-clock-time",
          "model_id": "claude-code-sonnet-4-6",
          "n": 100,
          "notes": "Printed Figure 2 label.",
          "status": "verified",
          "value": 1049.2
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "compbiobench-evidence-sonnet-results"
          ],
          "metric_id": "mean-cost",
          "model_id": "claude-code-sonnet-4-6",
          "n": 100,
          "notes": "Printed Figure 2 label.",
          "status": "verified",
          "value": 1.2
        }
      ],
      "scope": {
        "filter": "All 100 v1 tasks.",
        "n": 100,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact one-run Sonnet setting and Figure 2 values verified.",
        "status": "verified"
      },
      "work_id": "compbiobench-preprint",
      "work_version_id": "compbiobench-preprint-2026-04-09"
    },
    {
      "benchmark_id": "compbiobench",
      "benchmark_version": "v1",
      "comparability_group": "compbiobench-v1-hardest-sonnet-high-one-run",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "compbiobench-evidence-sonnet-hardest-protocol",
          "locator": {
            "note": "Exact Claude Code Sonnet configuration and common protocol.",
            "type": "page",
            "value": "PDF pp. 13–14, Methods"
          },
          "source_id": "compbiobench-preprint",
          "source_type": "work",
          "supports": [
            "/protocol"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "compbiobench-evidence-sonnet-hardest-results",
          "locator": {
            "note": "Levels 4 and 5 grouped; 17 tasks; Sonnet label 53.",
            "type": "figure",
            "value": "PDF p. 11, Supplementary Figure 2B and caption"
          },
          "source_id": "compbiobench-preprint",
          "source_type": "work",
          "supports": [
            "/scope",
            "/metrics",
            "/results"
          ]
        }
      ],
      "id": "compbiobench-sonnet-hardest",
      "metrics": [
        {
          "aggregation": "problem-weighted mean",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Accuracy — difficulty Levels 4–5",
          "tolerance": "whitespace-stripped exact match",
          "unit": "percent"
        }
      ],
      "model_ids": [
        "claude-code-sonnet-4-6"
      ],
      "protocol": {
        "contamination": {
          "notes": "No pretraining decontamination analysis.",
          "reporting_status": "reported",
          "value": "benchmark construction reduces lookup shortcuts"
        },
        "grader": {
          "human_review": true,
          "model": null,
          "reporting_status": "reported",
          "type": "whitespace-stripped exact string match with occasional manual format-only correction"
        },
        "reasoning": {
          "notes": "Claude Code --effort high.",
          "reporting_status": "reported",
          "value": "high"
        },
        "repeats": {
          "notes": "One run.",
          "reporting_status": "reported",
          "value": 1
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public wrapper prompt has no examples.",
          "reporting_status": "reported",
          "value": "zero-shot"
        },
        "statistical": {
          "notes": "No CI.",
          "reporting_status": "reported",
          "value": "problem-weighted accuracy in one run over the 17-task subset"
        },
        "system_prompt_public": {
          "notes": "Benchmark wrapper public; vendor system prompt not public.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Same run as full result.",
          "reporting_status": "reported",
          "value": "120 minutes initially; timeout rerun once with 240 minutes"
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "No distinct browser UI specified.",
            "reporting_status": "reported",
            "value": "web access through agent environment"
          },
          "code_execution": {
            "notes": "Auto-approved commands.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "Per-question Conda clone, not a hard container.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "External biological resources permitted.",
            "reporting_status": "reported",
            "value": true
          },
          "external_tools": {
            "notes": "Installation/retrieval allowed.",
            "reporting_status": "reported",
            "value": true
          },
          "internet": {
            "notes": "Web enabled.",
            "reporting_status": "reported",
            "value": true
          }
        },
        "turns": {
          "notes": "Claude Code agent.",
          "reporting_status": "reported",
          "value": "multi-turn agent trajectory"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "compbiobench-evidence-sonnet-hardest-results"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-code-sonnet-4-6",
          "n": 17,
          "notes": "Supplementary Figure 2 label; one run.",
          "status": "verified",
          "value": 53.0
        }
      ],
      "scope": {
        "filter": "Contributor-rated difficulty Levels 4 and 5 grouped together.",
        "n": 17,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "compbiobench-difficulty-levels-4-5",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Seventeen-task grouped scope and 53% one-run value verified.",
        "status": "verified"
      },
      "work_id": "compbiobench-preprint",
      "work_version_id": "compbiobench-preprint-2026-04-09"
    },
    {
      "benchmark_id": "flip-aav",
      "benchmark_version": "original-2021",
      "comparability_group": "flip-aav-des-mut-spearman",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "flip-aav-des-mut-scope-protocol",
          "locator": {
            "note": "Exact train/test counts, split rule, baseline identities, pooling, optimizer, batch sizes, early stopping, and hardware.",
            "type": "table",
            "value": "pp. 4–7, Table 2 and Sections 3–4"
          },
          "source_id": "flip-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/protocol",
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "flip-aav-des-mut-results",
          "locator": {
            "note": "Published held-out Spearman values; em dashes and NA entries are omitted rather than encoded as zeros.",
            "type": "table",
            "value": "p. 9, Table 5 (Des-Mut column)"
          },
          "source_id": "flip-paper",
          "source_type": "work",
          "supports": [
            "/metrics",
            "/results"
          ]
        }
      ],
      "id": "flip-aav-des-mut",
      "metrics": [
        {
          "aggregation": "computed across all examples in this held-out test split",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "spearman-correlation",
          "pass_threshold": null,
          "range": [
            -1,
            1
          ],
          "source_label": "Spearman correlation",
          "tolerance": null,
          "unit": "correlation"
        },
        {
          "aggregation": "mean across all examples in this held-out test split",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "mean-squared-error",
          "pass_threshold": null,
          "range": null,
          "source_label": "Mean squared error",
          "tolerance": null,
          "unit": "squared fitness units"
        }
      ],
      "model_ids": [
        "flip-cnn",
        "flip-esm-untrained-mean",
        "flip-esm-untrained-mut-mean",
        "flip-esm1b-mean",
        "flip-esm1b-mut-mean",
        "flip-esm1v-mean",
        "flip-esm1v-mut-mean",
        "flip-levenshtein",
        "flip-ridge"
      ],
      "protocol": {
        "contamination": {
          "notes": null,
          "reporting_status": "reported",
          "value": "Pool shift: model-designed variants are training data and sampled variants are held out."
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "deterministic regression scorer"
        },
        "reasoning": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "repeats": {
          "notes": "A common repeat count for the published table value is not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "A common random seed is not reported in the main paper.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "statistical": {
          "notes": "Mean-squared error is also evaluated in supplementary analyses and supported by the public scorer.",
          "reporting_status": "reported",
          "value": "Spearman correlation over held-out test examples; the main table reports point values without confidence intervals."
        },
        "system_prompt_public": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "temperature": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "time_budget": {
          "notes": "No common runtime budget; the paper reports Nvidia Quadro RTX A6000 hardware and omits selected infeasible per-AA runs.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "code_execution": {
            "notes": "Training code is part of the evaluation implementation, not an affordance granted to an agent.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "container": {
            "notes": "No common container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "external_tools": {
            "notes": "ESM representations are frozen; supervised heads or baselines train on the split.",
            "reporting_status": "reported",
            "value": "model-specific frozen embeddings or one-hot sequence encoding"
          },
          "internet": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          }
        },
        "turns": {
          "notes": "Batch supervised training and inference.",
          "reporting_status": "not_applicable",
          "value": "not-applicable"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-des-mut-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1b-mean",
          "n": 82583,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.59
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-des-mut-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1b-mut-mean",
          "n": 82583,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.7
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-des-mut-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1v-mean",
          "n": 82583,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.44
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-des-mut-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1v-mut-mean",
          "n": 82583,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.71
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-des-mut-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm-untrained-mean",
          "n": 82583,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.34
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-des-mut-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm-untrained-mut-mean",
          "n": 82583,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.64
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-des-mut-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-ridge",
          "n": 82583,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.53
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-des-mut-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-cnn",
          "n": 82583,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.75
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-des-mut-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-levenshtein",
          "n": 82583,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": -0.07
        }
      ],
      "scope": {
        "filter": "Train on 201,426 designed-pool variants; test on all 82,583 sampled-pool variants.",
        "n": 82583,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "aav-des-mut-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Creator-paper split setting and all published main-table Spearman values are registered.",
        "status": "verified"
      },
      "work_id": "flip-paper",
      "work_version_id": "flip-paper-2021-10-11"
    },
    {
      "benchmark_id": "flip-aav",
      "benchmark_version": "original-2021",
      "comparability_group": "flip-aav-low-vs-high-spearman",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "flip-aav-low-vs-high-scope-protocol",
          "locator": {
            "note": "Exact train/test counts, split rule, baseline identities, pooling, optimizer, batch sizes, early stopping, and hardware.",
            "type": "table",
            "value": "pp. 4–7, Table 2 and Sections 3–4"
          },
          "source_id": "flip-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/protocol",
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "flip-aav-low-vs-high-results",
          "locator": {
            "note": "Published held-out Spearman values; em dashes and NA entries are omitted rather than encoded as zeros.",
            "type": "table",
            "value": "p. 9, Table 5 (low-vs-high column)"
          },
          "source_id": "flip-paper",
          "source_type": "work",
          "supports": [
            "/metrics",
            "/results"
          ]
        }
      ],
      "id": "flip-aav-low-vs-high",
      "metrics": [
        {
          "aggregation": "computed across all examples in this held-out test split",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "spearman-correlation",
          "pass_threshold": null,
          "range": [
            -1,
            1
          ],
          "source_label": "Spearman correlation",
          "tolerance": null,
          "unit": "correlation"
        },
        {
          "aggregation": "mean across all examples in this held-out test split",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "mean-squared-error",
          "pass_threshold": null,
          "range": null,
          "source_label": "Mean squared error",
          "tolerance": null,
          "unit": "squared fitness units"
        }
      ],
      "model_ids": [
        "flip-cnn",
        "flip-esm-untrained-mean",
        "flip-esm-untrained-mut-mean",
        "flip-esm-untrained-per-aa",
        "flip-esm1b-mean",
        "flip-esm1b-mut-mean",
        "flip-esm1b-per-aa",
        "flip-esm1v-mean",
        "flip-esm1v-mut-mean",
        "flip-esm1v-per-aa",
        "flip-levenshtein",
        "flip-ridge"
      ],
      "protocol": {
        "contamination": {
          "notes": null,
          "reporting_status": "reported",
          "value": "Fitness extrapolation above wild-type fitness."
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "deterministic regression scorer"
        },
        "reasoning": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "repeats": {
          "notes": "A common repeat count for the published table value is not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "A common random seed is not reported in the main paper.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "statistical": {
          "notes": "Mean-squared error is also evaluated in supplementary analyses and supported by the public scorer.",
          "reporting_status": "reported",
          "value": "Spearman correlation over held-out test examples; the main table reports point values without confidence intervals."
        },
        "system_prompt_public": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "temperature": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "time_budget": {
          "notes": "No common runtime budget; the paper reports Nvidia Quadro RTX A6000 hardware and omits selected infeasible per-AA runs.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "code_execution": {
            "notes": "Training code is part of the evaluation implementation, not an affordance granted to an agent.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "container": {
            "notes": "No common container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "external_tools": {
            "notes": "ESM representations are frozen; supervised heads or baselines train on the split.",
            "reporting_status": "reported",
            "value": "model-specific frozen embeddings or one-hot sequence encoding"
          },
          "internet": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          }
        },
        "turns": {
          "notes": "Batch supervised training and inference.",
          "reporting_status": "not_applicable",
          "value": "not-applicable"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-low-vs-high-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1b-per-aa",
          "n": 35037,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.39
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-low-vs-high-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1b-mean",
          "n": 35037,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.18
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-low-vs-high-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1b-mut-mean",
          "n": 35037,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.33
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-low-vs-high-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1v-per-aa",
          "n": 35037,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.34
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-low-vs-high-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1v-mean",
          "n": 35037,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.2
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-low-vs-high-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1v-mut-mean",
          "n": 35037,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.31
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-low-vs-high-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm-untrained-per-aa",
          "n": 35037,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.08
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-low-vs-high-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm-untrained-mean",
          "n": 35037,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.22
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-low-vs-high-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm-untrained-mut-mean",
          "n": 35037,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.24
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-low-vs-high-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-ridge",
          "n": 35037,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.12
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-low-vs-high-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-cnn",
          "n": 35037,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.34
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-low-vs-high-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-levenshtein",
          "n": 35037,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.25
        }
      ],
      "scope": {
        "filter": "Train on 47,546 sampled-pool examples at or below wild-type fitness; test on 35,037 examples above wild-type fitness.",
        "n": 35037,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "aav-low-vs-high-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Creator-paper split setting and all published main-table Spearman values are registered.",
        "status": "verified"
      },
      "work_id": "flip-paper",
      "work_version_id": "flip-paper-2021-10-11"
    },
    {
      "benchmark_id": "flip-aav",
      "benchmark_version": "original-2021",
      "comparability_group": "flip-aav-mut-des-spearman",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "flip-aav-mut-des-scope-protocol",
          "locator": {
            "note": "Exact train/test counts, split rule, baseline identities, pooling, optimizer, batch sizes, early stopping, and hardware.",
            "type": "table",
            "value": "pp. 4–7, Table 2 and Sections 3–4"
          },
          "source_id": "flip-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/protocol",
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "flip-aav-mut-des-results",
          "locator": {
            "note": "Published held-out Spearman values; em dashes and NA entries are omitted rather than encoded as zeros.",
            "type": "table",
            "value": "p. 9, Table 5 (Mut-Des column)"
          },
          "source_id": "flip-paper",
          "source_type": "work",
          "supports": [
            "/metrics",
            "/results"
          ]
        }
      ],
      "id": "flip-aav-mut-des",
      "metrics": [
        {
          "aggregation": "computed across all examples in this held-out test split",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "spearman-correlation",
          "pass_threshold": null,
          "range": [
            -1,
            1
          ],
          "source_label": "Spearman correlation",
          "tolerance": null,
          "unit": "correlation"
        },
        {
          "aggregation": "mean across all examples in this held-out test split",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "mean-squared-error",
          "pass_threshold": null,
          "range": null,
          "source_label": "Mean squared error",
          "tolerance": null,
          "unit": "squared fitness units"
        }
      ],
      "model_ids": [
        "flip-cnn",
        "flip-esm-untrained-mean",
        "flip-esm-untrained-mut-mean",
        "flip-esm-untrained-per-aa",
        "flip-esm1b-mean",
        "flip-esm1b-mut-mean",
        "flip-esm1b-per-aa",
        "flip-esm1v-mean",
        "flip-esm1v-mut-mean",
        "flip-esm1v-per-aa",
        "flip-levenshtein",
        "flip-ridge"
      ],
      "protocol": {
        "contamination": {
          "notes": null,
          "reporting_status": "reported",
          "value": "Pool shift: sampled variants are training data and model-designed variants are held out."
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "deterministic regression scorer"
        },
        "reasoning": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "repeats": {
          "notes": "A common repeat count for the published table value is not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "A common random seed is not reported in the main paper.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "statistical": {
          "notes": "Mean-squared error is also evaluated in supplementary analyses and supported by the public scorer.",
          "reporting_status": "reported",
          "value": "Spearman correlation over held-out test examples; the main table reports point values without confidence intervals."
        },
        "system_prompt_public": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "temperature": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "time_budget": {
          "notes": "No common runtime budget; the paper reports Nvidia Quadro RTX A6000 hardware and omits selected infeasible per-AA runs.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "code_execution": {
            "notes": "Training code is part of the evaluation implementation, not an affordance granted to an agent.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "container": {
            "notes": "No common container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "external_tools": {
            "notes": "ESM representations are frozen; supervised heads or baselines train on the split.",
            "reporting_status": "reported",
            "value": "model-specific frozen embeddings or one-hot sequence encoding"
          },
          "internet": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          }
        },
        "turns": {
          "notes": "Batch supervised training and inference.",
          "reporting_status": "not_applicable",
          "value": "not-applicable"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-mut-des-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1b-per-aa",
          "n": 201426,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.76
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-mut-des-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1b-mean",
          "n": 201426,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.63
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-mut-des-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1b-mut-mean",
          "n": 201426,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.7
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-mut-des-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1v-per-aa",
          "n": 201426,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.79
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-mut-des-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1v-mean",
          "n": 201426,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.55
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-mut-des-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1v-mut-mean",
          "n": 201426,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.7
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-mut-des-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm-untrained-per-aa",
          "n": 201426,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.56
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-mut-des-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm-untrained-mean",
          "n": 201426,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.27
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-mut-des-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm-untrained-mut-mean",
          "n": 201426,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.62
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-mut-des-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-ridge",
          "n": 201426,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.64
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-mut-des-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-cnn",
          "n": 201426,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.71
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-mut-des-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-levenshtein",
          "n": 201426,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.6
        }
      ],
      "scope": {
        "filter": "Train on 82,583 sampled-pool variants; test on all 201,426 designed-pool variants.",
        "n": 201426,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "aav-mut-des-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Creator-paper split setting and all published main-table Spearman values are registered.",
        "status": "verified"
      },
      "work_id": "flip-paper",
      "work_version_id": "flip-paper-2021-10-11"
    },
    {
      "benchmark_id": "flip-aav",
      "benchmark_version": "original-2021",
      "comparability_group": "flip-aav-one-vs-rest-spearman",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "flip-aav-one-vs-rest-scope-protocol",
          "locator": {
            "note": "Exact train/test counts, split rule, baseline identities, pooling, optimizer, batch sizes, early stopping, and hardware.",
            "type": "table",
            "value": "pp. 4–7, Table 2 and Sections 3–4"
          },
          "source_id": "flip-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/protocol",
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "flip-aav-one-vs-rest-results",
          "locator": {
            "note": "Published held-out Spearman values; em dashes and NA entries are omitted rather than encoded as zeros.",
            "type": "table",
            "value": "p. 9, Table 5 (1-vs-rest column)"
          },
          "source_id": "flip-paper",
          "source_type": "work",
          "supports": [
            "/metrics",
            "/results"
          ]
        }
      ],
      "id": "flip-aav-one-vs-rest",
      "metrics": [
        {
          "aggregation": "computed across all examples in this held-out test split",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "spearman-correlation",
          "pass_threshold": null,
          "range": [
            -1,
            1
          ],
          "source_label": "Spearman correlation",
          "tolerance": null,
          "unit": "correlation"
        },
        {
          "aggregation": "mean across all examples in this held-out test split",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "mean-squared-error",
          "pass_threshold": null,
          "range": null,
          "source_label": "Mean squared error",
          "tolerance": null,
          "unit": "squared fitness units"
        }
      ],
      "model_ids": [
        "flip-cnn",
        "flip-esm-untrained-mean",
        "flip-esm-untrained-mut-mean",
        "flip-esm-untrained-per-aa",
        "flip-esm1b-mean",
        "flip-esm1b-mut-mean",
        "flip-esm1b-per-aa",
        "flip-esm1v-mean",
        "flip-esm1v-mut-mean",
        "flip-esm1v-per-aa",
        "flip-levenshtein",
        "flip-ridge"
      ],
      "protocol": {
        "contamination": {
          "notes": null,
          "reporting_status": "reported",
          "value": "Mutation-depth extrapolation; the held-out set excludes wild type and single-mutant training examples."
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "deterministic regression scorer"
        },
        "reasoning": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "repeats": {
          "notes": "A common repeat count for the published table value is not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "A common random seed is not reported in the main paper.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "statistical": {
          "notes": "Mean-squared error is also evaluated in supplementary analyses and supported by the public scorer.",
          "reporting_status": "reported",
          "value": "Spearman correlation over held-out test examples; the main table reports point values without confidence intervals."
        },
        "system_prompt_public": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "temperature": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "time_budget": {
          "notes": "No common runtime budget; the paper reports Nvidia Quadro RTX A6000 hardware and omits selected infeasible per-AA runs.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "code_execution": {
            "notes": "Training code is part of the evaluation implementation, not an affordance granted to an agent.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "container": {
            "notes": "No common container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "external_tools": {
            "notes": "ESM representations are frozen; supervised heads or baselines train on the split.",
            "reporting_status": "reported",
            "value": "model-specific frozen embeddings or one-hot sequence encoding"
          },
          "internet": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          }
        },
        "turns": {
          "notes": "Batch supervised training and inference.",
          "reporting_status": "not_applicable",
          "value": "not-applicable"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-one-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1b-per-aa",
          "n": 81413,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.03
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-one-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1b-mean",
          "n": 81413,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.04
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-one-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1b-mut-mean",
          "n": 81413,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.31
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-one-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1v-per-aa",
          "n": 81413,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.1
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-one-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1v-mean",
          "n": 81413,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.18
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-one-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1v-mut-mean",
          "n": 81413,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.44
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-one-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm-untrained-per-aa",
          "n": 81413,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.18
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-one-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm-untrained-mean",
          "n": 81413,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.01
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-one-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm-untrained-mut-mean",
          "n": 81413,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.26
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-one-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-ridge",
          "n": 81413,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.22
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-one-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-cnn",
          "n": 81413,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.48
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-one-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-levenshtein",
          "n": 81413,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": -0.11
        }
      ],
      "scope": {
        "filter": "Train on wild type and single mutants (1,170 examples); test on the remaining 81,413 sampled-pool examples.",
        "n": 81413,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "aav-one-vs-rest-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Creator-paper split setting and all published main-table Spearman values are registered.",
        "status": "verified"
      },
      "work_id": "flip-paper",
      "work_version_id": "flip-paper-2021-10-11"
    },
    {
      "benchmark_id": "flip-aav",
      "benchmark_version": "original-2021",
      "comparability_group": "flip-aav-sampled-spearman",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "flip-aav-sampled-scope-protocol",
          "locator": {
            "note": "Exact train/test counts, split rule, baseline identities, pooling, optimizer, batch sizes, early stopping, and hardware.",
            "type": "table",
            "value": "pp. 4–7, Table 2 and Sections 3–4"
          },
          "source_id": "flip-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/protocol",
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "flip-aav-sampled-results",
          "locator": {
            "note": "Published held-out Spearman values; em dashes and NA entries are omitted rather than encoded as zeros.",
            "type": "table",
            "value": "p. 10, Table 7 (AAV column)"
          },
          "source_id": "flip-paper",
          "source_type": "work",
          "supports": [
            "/metrics",
            "/results"
          ]
        }
      ],
      "id": "flip-aav-sampled",
      "metrics": [
        {
          "aggregation": "computed across all examples in this held-out test split",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "spearman-correlation",
          "pass_threshold": null,
          "range": [
            -1,
            1
          ],
          "source_label": "Spearman correlation",
          "tolerance": null,
          "unit": "correlation"
        },
        {
          "aggregation": "mean across all examples in this held-out test split",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "mean-squared-error",
          "pass_threshold": null,
          "range": null,
          "source_label": "Mean squared error",
          "tolerance": null,
          "unit": "squared fitness units"
        }
      ],
      "model_ids": [
        "flip-cnn",
        "flip-esm-untrained-per-aa",
        "flip-esm1b-per-aa",
        "flip-esm1v-per-aa",
        "flip-ridge"
      ],
      "protocol": {
        "contamination": {
          "notes": null,
          "reporting_status": "reported",
          "value": "Random split can place closely related variants across train/test and is explicitly described as optimistic."
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "deterministic regression scorer"
        },
        "reasoning": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "repeats": {
          "notes": "A common repeat count for the published table value is not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "A common random seed is not reported in the main paper.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "statistical": {
          "notes": "Mean-squared error is also evaluated in supplementary analyses and supported by the public scorer.",
          "reporting_status": "reported",
          "value": "Spearman correlation over held-out test examples; the main table reports point values without confidence intervals."
        },
        "system_prompt_public": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "temperature": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "time_budget": {
          "notes": "No common runtime budget; the paper reports Nvidia Quadro RTX A6000 hardware and omits selected infeasible per-AA runs.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "code_execution": {
            "notes": "Training code is part of the evaluation implementation, not an affordance granted to an agent.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "container": {
            "notes": "No common container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "external_tools": {
            "notes": "ESM representations are frozen; supervised heads or baselines train on the split.",
            "reporting_status": "reported",
            "value": "model-specific frozen embeddings or one-hot sequence encoding"
          },
          "internet": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          }
        },
        "turns": {
          "notes": "Batch supervised training and inference.",
          "reporting_status": "not_applicable",
          "value": "not-applicable"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-sampled-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1b-per-aa",
          "n": 16517,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.9
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-sampled-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1v-per-aa",
          "n": 16517,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.92
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-sampled-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm-untrained-per-aa",
          "n": 16517,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.78
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-sampled-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-ridge",
          "n": 16517,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.83
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-sampled-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-cnn",
          "n": 16517,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.92
        }
      ],
      "scope": {
        "filter": "Random 80/20 sampled-pool split: 66,066 train and 16,517 test; creator marks it discourse-only and unsuitable for performance comparison.",
        "n": 16517,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "aav-sampled-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Creator-paper discourse-only random split; published results are retained but isolated from active comparison groups.",
        "status": "verified"
      },
      "work_id": "flip-paper",
      "work_version_id": "flip-paper-2021-10-11"
    },
    {
      "benchmark_id": "flip-aav",
      "benchmark_version": "original-2021",
      "comparability_group": "flip-aav-seven-vs-rest-spearman",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "flip-aav-seven-vs-rest-scope-protocol",
          "locator": {
            "note": "Exact train/test counts, split rule, baseline identities, pooling, optimizer, batch sizes, early stopping, and hardware.",
            "type": "table",
            "value": "pp. 4–7, Table 2 and Sections 3–4"
          },
          "source_id": "flip-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/protocol",
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "flip-aav-seven-vs-rest-results",
          "locator": {
            "note": "Published held-out Spearman values; em dashes and NA entries are omitted rather than encoded as zeros.",
            "type": "table",
            "value": "p. 9, Table 5 (7-vs-rest column)"
          },
          "source_id": "flip-paper",
          "source_type": "work",
          "supports": [
            "/metrics",
            "/results"
          ]
        }
      ],
      "id": "flip-aav-seven-vs-rest",
      "metrics": [
        {
          "aggregation": "computed across all examples in this held-out test split",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "spearman-correlation",
          "pass_threshold": null,
          "range": [
            -1,
            1
          ],
          "source_label": "Spearman correlation",
          "tolerance": null,
          "unit": "correlation"
        },
        {
          "aggregation": "mean across all examples in this held-out test split",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "mean-squared-error",
          "pass_threshold": null,
          "range": null,
          "source_label": "Mean squared error",
          "tolerance": null,
          "unit": "squared fitness units"
        }
      ],
      "model_ids": [
        "flip-cnn",
        "flip-esm-untrained-mean",
        "flip-esm-untrained-mut-mean",
        "flip-esm-untrained-per-aa",
        "flip-esm1b-mean",
        "flip-esm1b-mut-mean",
        "flip-esm1b-per-aa",
        "flip-esm1v-mean",
        "flip-esm1v-mut-mean",
        "flip-esm1v-per-aa",
        "flip-levenshtein",
        "flip-ridge"
      ],
      "protocol": {
        "contamination": {
          "notes": null,
          "reporting_status": "reported",
          "value": "Mutation-depth extrapolation above seven changes."
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "deterministic regression scorer"
        },
        "reasoning": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "repeats": {
          "notes": "A common repeat count for the published table value is not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "A common random seed is not reported in the main paper.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "statistical": {
          "notes": "Mean-squared error is also evaluated in supplementary analyses and supported by the public scorer.",
          "reporting_status": "reported",
          "value": "Spearman correlation over held-out test examples; the main table reports point values without confidence intervals."
        },
        "system_prompt_public": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "temperature": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "time_budget": {
          "notes": "No common runtime budget; the paper reports Nvidia Quadro RTX A6000 hardware and omits selected infeasible per-AA runs.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "code_execution": {
            "notes": "Training code is part of the evaluation implementation, not an affordance granted to an agent.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "container": {
            "notes": "No common container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "external_tools": {
            "notes": "ESM representations are frozen; supervised heads or baselines train on the split.",
            "reporting_status": "reported",
            "value": "model-specific frozen embeddings or one-hot sequence encoding"
          },
          "internet": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          }
        },
        "turns": {
          "notes": "Batch supervised training and inference.",
          "reporting_status": "not_applicable",
          "value": "not-applicable"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-seven-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1b-per-aa",
          "n": 12581,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.65
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-seven-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1b-mean",
          "n": 12581,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.46
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-seven-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1b-mut-mean",
          "n": 12581,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.61
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-seven-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1v-per-aa",
          "n": 12581,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.7
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-seven-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1v-mean",
          "n": 12581,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.45
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-seven-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1v-mut-mean",
          "n": 12581,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.64
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-seven-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm-untrained-per-aa",
          "n": 12581,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.42
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-seven-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm-untrained-mean",
          "n": 12581,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.22
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-seven-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm-untrained-mut-mean",
          "n": 12581,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.56
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-seven-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-ridge",
          "n": 12581,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.65
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-seven-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-cnn",
          "n": 12581,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.74
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-seven-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-levenshtein",
          "n": 12581,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.53
        }
      ],
      "scope": {
        "filter": "Train on sampled-pool variants with up to seven changes (70,002 examples); test on 12,581 variants with more changes.",
        "n": 12581,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "aav-seven-vs-rest-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Creator-paper split setting and all published main-table Spearman values are registered.",
        "status": "verified"
      },
      "work_id": "flip-paper",
      "work_version_id": "flip-paper-2021-10-11"
    },
    {
      "benchmark_id": "flip-aav",
      "benchmark_version": "original-2021",
      "comparability_group": "flip-aav-two-vs-rest-spearman",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "flip-aav-two-vs-rest-scope-protocol",
          "locator": {
            "note": "Exact train/test counts, split rule, baseline identities, pooling, optimizer, batch sizes, early stopping, and hardware.",
            "type": "table",
            "value": "pp. 4–7, Table 2 and Sections 3–4"
          },
          "source_id": "flip-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/protocol",
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "flip-aav-two-vs-rest-results",
          "locator": {
            "note": "Published held-out Spearman values; em dashes and NA entries are omitted rather than encoded as zeros.",
            "type": "table",
            "value": "p. 9, Table 5 (2-vs-rest column)"
          },
          "source_id": "flip-paper",
          "source_type": "work",
          "supports": [
            "/metrics",
            "/results"
          ]
        }
      ],
      "id": "flip-aav-two-vs-rest",
      "metrics": [
        {
          "aggregation": "computed across all examples in this held-out test split",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "spearman-correlation",
          "pass_threshold": null,
          "range": [
            -1,
            1
          ],
          "source_label": "Spearman correlation",
          "tolerance": null,
          "unit": "correlation"
        },
        {
          "aggregation": "mean across all examples in this held-out test split",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "mean-squared-error",
          "pass_threshold": null,
          "range": null,
          "source_label": "Mean squared error",
          "tolerance": null,
          "unit": "squared fitness units"
        }
      ],
      "model_ids": [
        "flip-cnn",
        "flip-esm-untrained-mean",
        "flip-esm-untrained-mut-mean",
        "flip-esm-untrained-per-aa",
        "flip-esm1b-mean",
        "flip-esm1b-mut-mean",
        "flip-esm1b-per-aa",
        "flip-esm1v-mean",
        "flip-esm1v-mut-mean",
        "flip-esm1v-per-aa",
        "flip-levenshtein",
        "flip-ridge"
      ],
      "protocol": {
        "contamination": {
          "notes": null,
          "reporting_status": "reported",
          "value": "Mutation-depth extrapolation above two changes."
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "deterministic regression scorer"
        },
        "reasoning": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "repeats": {
          "notes": "A common repeat count for the published table value is not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "A common random seed is not reported in the main paper.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "statistical": {
          "notes": "Mean-squared error is also evaluated in supplementary analyses and supported by the public scorer.",
          "reporting_status": "reported",
          "value": "Spearman correlation over held-out test examples; the main table reports point values without confidence intervals."
        },
        "system_prompt_public": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "temperature": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "time_budget": {
          "notes": "No common runtime budget; the paper reports Nvidia Quadro RTX A6000 hardware and omits selected infeasible per-AA runs.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "code_execution": {
            "notes": "Training code is part of the evaluation implementation, not an affordance granted to an agent.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "container": {
            "notes": "No common container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "external_tools": {
            "notes": "ESM representations are frozen; supervised heads or baselines train on the split.",
            "reporting_status": "reported",
            "value": "model-specific frozen embeddings or one-hot sequence encoding"
          },
          "internet": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          }
        },
        "turns": {
          "notes": "Batch supervised training and inference.",
          "reporting_status": "not_applicable",
          "value": "not-applicable"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-two-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1b-per-aa",
          "n": 50776,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.65
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-two-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1b-mean",
          "n": 50776,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.26
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-two-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1b-mut-mean",
          "n": 50776,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.65
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-two-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1v-per-aa",
          "n": 50776,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.7
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-two-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1v-mean",
          "n": 50776,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.16
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-two-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1v-mut-mean",
          "n": 50776,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.64
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-two-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm-untrained-per-aa",
          "n": 50776,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.22
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-two-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm-untrained-mean",
          "n": 50776,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.14
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-two-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm-untrained-mut-mean",
          "n": 50776,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.16
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-two-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-ridge",
          "n": 50776,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.03
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-two-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-cnn",
          "n": 50776,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.74
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-aav-two-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-levenshtein",
          "n": 50776,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.57
        }
      ],
      "scope": {
        "filter": "Train on wild type, single, and double mutants (31,807 examples); test on 50,776 higher-mutation examples.",
        "n": 50776,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "aav-two-vs-rest-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Creator-paper split setting and all published main-table Spearman values are registered.",
        "status": "verified"
      },
      "work_id": "flip-paper",
      "work_version_id": "flip-paper-2021-10-11"
    },
    {
      "benchmark_id": "flip-gb1",
      "benchmark_version": "original-2021",
      "comparability_group": "flip-gb1-low-vs-high-spearman",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "flip-gb1-low-vs-high-scope-protocol",
          "locator": {
            "note": "Exact train/test counts, split rule, baseline identities, pooling, optimizer, batch sizes, early stopping, and hardware.",
            "type": "table",
            "value": "pp. 4–7, Table 2 and Sections 3–4"
          },
          "source_id": "flip-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/protocol",
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "flip-gb1-low-vs-high-results",
          "locator": {
            "note": "Published held-out Spearman values; em dashes and NA entries are omitted rather than encoded as zeros.",
            "type": "table",
            "value": "p. 8, Table 4 (low-vs-high column)"
          },
          "source_id": "flip-paper",
          "source_type": "work",
          "supports": [
            "/metrics",
            "/results"
          ]
        }
      ],
      "id": "flip-gb1-low-vs-high",
      "metrics": [
        {
          "aggregation": "computed across all examples in this held-out test split",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "spearman-correlation",
          "pass_threshold": null,
          "range": [
            -1,
            1
          ],
          "source_label": "Spearman correlation",
          "tolerance": null,
          "unit": "correlation"
        },
        {
          "aggregation": "mean across all examples in this held-out test split",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "mean-squared-error",
          "pass_threshold": null,
          "range": null,
          "source_label": "Mean squared error",
          "tolerance": null,
          "unit": "squared fitness units"
        }
      ],
      "model_ids": [
        "flip-blosum62",
        "flip-cnn",
        "flip-esm-untrained-mean",
        "flip-esm-untrained-mut-mean",
        "flip-esm-untrained-per-aa",
        "flip-esm1b-mean",
        "flip-esm1b-mut-mean",
        "flip-esm1b-per-aa",
        "flip-esm1v-mean",
        "flip-esm1v-mut-mean",
        "flip-esm1v-per-aa",
        "flip-levenshtein",
        "flip-ridge"
      ],
      "protocol": {
        "contamination": {
          "notes": null,
          "reporting_status": "reported",
          "value": "Fitness extrapolation above wild-type fitness."
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "deterministic regression scorer"
        },
        "reasoning": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "repeats": {
          "notes": "A common repeat count for the published table value is not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "A common random seed is not reported in the main paper.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "statistical": {
          "notes": "Mean-squared error is also evaluated in supplementary analyses and supported by the public scorer.",
          "reporting_status": "reported",
          "value": "Spearman correlation over held-out test examples; the main table reports point values without confidence intervals."
        },
        "system_prompt_public": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "temperature": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "time_budget": {
          "notes": "No common runtime budget; the paper reports Nvidia Quadro RTX A6000 hardware and omits selected infeasible per-AA runs.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "code_execution": {
            "notes": "Training code is part of the evaluation implementation, not an affordance granted to an agent.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "container": {
            "notes": "No common container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "external_tools": {
            "notes": "ESM representations are frozen; supervised heads or baselines train on the split.",
            "reporting_status": "reported",
            "value": "model-specific frozen embeddings or one-hot sequence encoding"
          },
          "internet": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          }
        },
        "turns": {
          "notes": "Batch supervised training and inference.",
          "reporting_status": "not_applicable",
          "value": "not-applicable"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-low-vs-high-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1b-per-aa",
          "n": 3644,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.59
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-low-vs-high-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1b-mean",
          "n": 3644,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.13
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-low-vs-high-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1b-mut-mean",
          "n": 3644,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.45
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-low-vs-high-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1v-per-aa",
          "n": 3644,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.51
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-low-vs-high-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1v-mean",
          "n": 3644,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.1
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-low-vs-high-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1v-mut-mean",
          "n": 3644,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.49
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-low-vs-high-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm-untrained-per-aa",
          "n": 3644,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.23
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-low-vs-high-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm-untrained-mean",
          "n": 3644,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.1
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-low-vs-high-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm-untrained-mut-mean",
          "n": 3644,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.13
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-low-vs-high-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-ridge",
          "n": 3644,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.34
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-low-vs-high-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-cnn",
          "n": 3644,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.51
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-low-vs-high-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-levenshtein",
          "n": 3644,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": -0.1
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-low-vs-high-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-blosum62",
          "n": 3644,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": -0.13
        }
      ],
      "scope": {
        "filter": "Train on 5,089 examples at or below wild-type fitness; test on 3,644 examples above wild-type fitness.",
        "n": 3644,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "gb1-low-vs-high-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Creator-paper split setting and all published main-table Spearman values are registered.",
        "status": "verified"
      },
      "work_id": "flip-paper",
      "work_version_id": "flip-paper-2021-10-11"
    },
    {
      "benchmark_id": "flip-gb1",
      "benchmark_version": "original-2021",
      "comparability_group": "flip-gb1-one-vs-rest-spearman",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "flip-gb1-one-vs-rest-scope-protocol",
          "locator": {
            "note": "Exact train/test counts, split rule, baseline identities, pooling, optimizer, batch sizes, early stopping, and hardware.",
            "type": "table",
            "value": "pp. 4–7, Table 2 and Sections 3–4"
          },
          "source_id": "flip-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/protocol",
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "flip-gb1-one-vs-rest-results",
          "locator": {
            "note": "Published held-out Spearman values; em dashes and NA entries are omitted rather than encoded as zeros.",
            "type": "table",
            "value": "p. 8, Table 4 (1-vs-rest column)"
          },
          "source_id": "flip-paper",
          "source_type": "work",
          "supports": [
            "/metrics",
            "/results"
          ]
        }
      ],
      "id": "flip-gb1-one-vs-rest",
      "metrics": [
        {
          "aggregation": "computed across all examples in this held-out test split",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "spearman-correlation",
          "pass_threshold": null,
          "range": [
            -1,
            1
          ],
          "source_label": "Spearman correlation",
          "tolerance": null,
          "unit": "correlation"
        },
        {
          "aggregation": "mean across all examples in this held-out test split",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "mean-squared-error",
          "pass_threshold": null,
          "range": null,
          "source_label": "Mean squared error",
          "tolerance": null,
          "unit": "squared fitness units"
        }
      ],
      "model_ids": [
        "flip-blosum62",
        "flip-cnn",
        "flip-esm-untrained-mean",
        "flip-esm-untrained-mut-mean",
        "flip-esm-untrained-per-aa",
        "flip-esm1b-mean",
        "flip-esm1b-mut-mean",
        "flip-esm1b-per-aa",
        "flip-esm1v-mean",
        "flip-esm1v-mut-mean",
        "flip-esm1v-per-aa",
        "flip-levenshtein",
        "flip-ridge"
      ],
      "protocol": {
        "contamination": {
          "notes": null,
          "reporting_status": "reported",
          "value": "Mutation-depth extrapolation from wild type and single mutants."
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "deterministic regression scorer"
        },
        "reasoning": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "repeats": {
          "notes": "A common repeat count for the published table value is not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "A common random seed is not reported in the main paper.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "statistical": {
          "notes": "Mean-squared error is also evaluated in supplementary analyses and supported by the public scorer.",
          "reporting_status": "reported",
          "value": "Spearman correlation over held-out test examples; the main table reports point values without confidence intervals."
        },
        "system_prompt_public": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "temperature": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "time_budget": {
          "notes": "No common runtime budget; the paper reports Nvidia Quadro RTX A6000 hardware and omits selected infeasible per-AA runs.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "code_execution": {
            "notes": "Training code is part of the evaluation implementation, not an affordance granted to an agent.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "container": {
            "notes": "No common container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "external_tools": {
            "notes": "ESM representations are frozen; supervised heads or baselines train on the split.",
            "reporting_status": "reported",
            "value": "model-specific frozen embeddings or one-hot sequence encoding"
          },
          "internet": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          }
        },
        "turns": {
          "notes": "Batch supervised training and inference.",
          "reporting_status": "not_applicable",
          "value": "not-applicable"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-one-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1b-per-aa",
          "n": 8704,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.28
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-one-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1b-mean",
          "n": 8704,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.32
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-one-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1b-mut-mean",
          "n": 8704,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": -0.08
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-one-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1v-per-aa",
          "n": 8704,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.28
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-one-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1v-mean",
          "n": 8704,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.32
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-one-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1v-mut-mean",
          "n": 8704,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.19
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-one-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm-untrained-per-aa",
          "n": 8704,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.06
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-one-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm-untrained-mean",
          "n": 8704,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.05
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-one-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm-untrained-mut-mean",
          "n": 8704,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.21
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-one-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-ridge",
          "n": 8704,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.28
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-one-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-cnn",
          "n": 8704,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.17
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-one-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-levenshtein",
          "n": 8704,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.17
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-one-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-blosum62",
          "n": 8704,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.15
        }
      ],
      "scope": {
        "filter": "Train on wild type and single mutants (29 examples); test on the remaining 8,704 variants.",
        "n": 8704,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "gb1-one-vs-rest-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Creator-paper split setting and all published main-table Spearman values are registered.",
        "status": "verified"
      },
      "work_id": "flip-paper",
      "work_version_id": "flip-paper-2021-10-11"
    },
    {
      "benchmark_id": "flip-gb1",
      "benchmark_version": "original-2021",
      "comparability_group": "flip-gb1-sampled-spearman",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "flip-gb1-sampled-scope-protocol",
          "locator": {
            "note": "Exact train/test counts, split rule, baseline identities, pooling, optimizer, batch sizes, early stopping, and hardware.",
            "type": "table",
            "value": "pp. 4–7, Table 2 and Sections 3–4"
          },
          "source_id": "flip-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/protocol",
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "flip-gb1-sampled-results",
          "locator": {
            "note": "Published held-out Spearman values; em dashes and NA entries are omitted rather than encoded as zeros.",
            "type": "table",
            "value": "p. 10, Table 7 (GB1 column)"
          },
          "source_id": "flip-paper",
          "source_type": "work",
          "supports": [
            "/metrics",
            "/results"
          ]
        }
      ],
      "id": "flip-gb1-sampled",
      "metrics": [
        {
          "aggregation": "computed across all examples in this held-out test split",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "spearman-correlation",
          "pass_threshold": null,
          "range": [
            -1,
            1
          ],
          "source_label": "Spearman correlation",
          "tolerance": null,
          "unit": "correlation"
        },
        {
          "aggregation": "mean across all examples in this held-out test split",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "mean-squared-error",
          "pass_threshold": null,
          "range": null,
          "source_label": "Mean squared error",
          "tolerance": null,
          "unit": "squared fitness units"
        }
      ],
      "model_ids": [
        "flip-cnn",
        "flip-esm-untrained-per-aa",
        "flip-esm1b-per-aa",
        "flip-esm1v-per-aa",
        "flip-ridge"
      ],
      "protocol": {
        "contamination": {
          "notes": null,
          "reporting_status": "reported",
          "value": "Random split can place closely related variants across train/test and is explicitly described as optimistic."
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "deterministic regression scorer"
        },
        "reasoning": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "repeats": {
          "notes": "A common repeat count for the published table value is not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "A common random seed is not reported in the main paper.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "statistical": {
          "notes": "Mean-squared error is also evaluated in supplementary analyses and supported by the public scorer.",
          "reporting_status": "reported",
          "value": "Spearman correlation over held-out test examples; the main table reports point values without confidence intervals."
        },
        "system_prompt_public": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "temperature": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "time_budget": {
          "notes": "No common runtime budget; the paper reports Nvidia Quadro RTX A6000 hardware and omits selected infeasible per-AA runs.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "code_execution": {
            "notes": "Training code is part of the evaluation implementation, not an affordance granted to an agent.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "container": {
            "notes": "No common container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "external_tools": {
            "notes": "ESM representations are frozen; supervised heads or baselines train on the split.",
            "reporting_status": "reported",
            "value": "model-specific frozen embeddings or one-hot sequence encoding"
          },
          "internet": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          }
        },
        "turns": {
          "notes": "Batch supervised training and inference.",
          "reporting_status": "not_applicable",
          "value": "not-applicable"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-sampled-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1b-per-aa",
          "n": 1772,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.92
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-sampled-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1v-per-aa",
          "n": 1772,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.92
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-sampled-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm-untrained-per-aa",
          "n": 1772,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.79
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-sampled-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-ridge",
          "n": 1772,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.82
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-sampled-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-cnn",
          "n": 1772,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.91
        }
      ],
      "scope": {
        "filter": "Random 80/20 split: 6,961 train and 1,772 test; creator marks it discourse-only and unsuitable for performance comparison.",
        "n": 1772,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "gb1-sampled-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Creator-paper discourse-only random split; published results are retained but isolated from active comparison groups.",
        "status": "verified"
      },
      "work_id": "flip-paper",
      "work_version_id": "flip-paper-2021-10-11"
    },
    {
      "benchmark_id": "flip-gb1",
      "benchmark_version": "original-2021",
      "comparability_group": "flip-gb1-three-vs-rest-spearman",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "flip-gb1-three-vs-rest-scope-protocol",
          "locator": {
            "note": "Exact train/test counts, split rule, baseline identities, pooling, optimizer, batch sizes, early stopping, and hardware.",
            "type": "table",
            "value": "pp. 4–7, Table 2 and Sections 3–4"
          },
          "source_id": "flip-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/protocol",
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "flip-gb1-three-vs-rest-results",
          "locator": {
            "note": "Published held-out Spearman values; em dashes and NA entries are omitted rather than encoded as zeros.",
            "type": "table",
            "value": "p. 8, Table 4 (3-vs-rest column)"
          },
          "source_id": "flip-paper",
          "source_type": "work",
          "supports": [
            "/metrics",
            "/results"
          ]
        }
      ],
      "id": "flip-gb1-three-vs-rest",
      "metrics": [
        {
          "aggregation": "computed across all examples in this held-out test split",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "spearman-correlation",
          "pass_threshold": null,
          "range": [
            -1,
            1
          ],
          "source_label": "Spearman correlation",
          "tolerance": null,
          "unit": "correlation"
        },
        {
          "aggregation": "mean across all examples in this held-out test split",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "mean-squared-error",
          "pass_threshold": null,
          "range": null,
          "source_label": "Mean squared error",
          "tolerance": null,
          "unit": "squared fitness units"
        }
      ],
      "model_ids": [
        "flip-blosum62",
        "flip-cnn",
        "flip-esm-untrained-mean",
        "flip-esm-untrained-mut-mean",
        "flip-esm-untrained-per-aa",
        "flip-esm1b-mean",
        "flip-esm1b-mut-mean",
        "flip-esm1b-per-aa",
        "flip-esm1v-mean",
        "flip-esm1v-mut-mean",
        "flip-esm1v-per-aa",
        "flip-levenshtein",
        "flip-ridge"
      ],
      "protocol": {
        "contamination": {
          "notes": null,
          "reporting_status": "reported",
          "value": "Mutation-depth extrapolation to four simultaneous mutations."
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "deterministic regression scorer"
        },
        "reasoning": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "repeats": {
          "notes": "A common repeat count for the published table value is not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "A common random seed is not reported in the main paper.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "statistical": {
          "notes": "Mean-squared error is also evaluated in supplementary analyses and supported by the public scorer.",
          "reporting_status": "reported",
          "value": "Spearman correlation over held-out test examples; the main table reports point values without confidence intervals."
        },
        "system_prompt_public": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "temperature": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "time_budget": {
          "notes": "No common runtime budget; the paper reports Nvidia Quadro RTX A6000 hardware and omits selected infeasible per-AA runs.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "code_execution": {
            "notes": "Training code is part of the evaluation implementation, not an affordance granted to an agent.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "container": {
            "notes": "No common container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "external_tools": {
            "notes": "ESM representations are frozen; supervised heads or baselines train on the split.",
            "reporting_status": "reported",
            "value": "model-specific frozen embeddings or one-hot sequence encoding"
          },
          "internet": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          }
        },
        "turns": {
          "notes": "Batch supervised training and inference.",
          "reporting_status": "not_applicable",
          "value": "not-applicable"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-three-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1b-per-aa",
          "n": 5765,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.79
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-three-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1b-mean",
          "n": 5765,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.54
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-three-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1b-mut-mean",
          "n": 5765,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.49
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-three-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1v-per-aa",
          "n": 5765,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.82
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-three-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1v-mean",
          "n": 5765,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.77
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-three-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1v-mut-mean",
          "n": 5765,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.8
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-three-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm-untrained-per-aa",
          "n": 5765,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.48
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-three-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm-untrained-mean",
          "n": 5765,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.46
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-three-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm-untrained-mut-mean",
          "n": 5765,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.57
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-three-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-ridge",
          "n": 5765,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.76
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-three-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-cnn",
          "n": 5765,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.83
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-three-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-levenshtein",
          "n": 5765,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": -0.04
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-three-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-blosum62",
          "n": 5765,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.01
        }
      ],
      "scope": {
        "filter": "Train through triple mutants (2,968 examples); test on 5,765 quadruple variants.",
        "n": 5765,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "gb1-three-vs-rest-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Creator-paper split setting and all published main-table Spearman values are registered.",
        "status": "verified"
      },
      "work_id": "flip-paper",
      "work_version_id": "flip-paper-2021-10-11"
    },
    {
      "benchmark_id": "flip-gb1",
      "benchmark_version": "original-2021",
      "comparability_group": "flip-gb1-two-vs-rest-spearman",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "flip-gb1-two-vs-rest-scope-protocol",
          "locator": {
            "note": "Exact train/test counts, split rule, baseline identities, pooling, optimizer, batch sizes, early stopping, and hardware.",
            "type": "table",
            "value": "pp. 4–7, Table 2 and Sections 3–4"
          },
          "source_id": "flip-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/protocol",
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "flip-gb1-two-vs-rest-results",
          "locator": {
            "note": "Published held-out Spearman values; em dashes and NA entries are omitted rather than encoded as zeros.",
            "type": "table",
            "value": "p. 8, Table 4 (2-vs-rest column)"
          },
          "source_id": "flip-paper",
          "source_type": "work",
          "supports": [
            "/metrics",
            "/results"
          ]
        }
      ],
      "id": "flip-gb1-two-vs-rest",
      "metrics": [
        {
          "aggregation": "computed across all examples in this held-out test split",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "spearman-correlation",
          "pass_threshold": null,
          "range": [
            -1,
            1
          ],
          "source_label": "Spearman correlation",
          "tolerance": null,
          "unit": "correlation"
        },
        {
          "aggregation": "mean across all examples in this held-out test split",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "mean-squared-error",
          "pass_threshold": null,
          "range": null,
          "source_label": "Mean squared error",
          "tolerance": null,
          "unit": "squared fitness units"
        }
      ],
      "model_ids": [
        "flip-blosum62",
        "flip-cnn",
        "flip-esm-untrained-mean",
        "flip-esm-untrained-mut-mean",
        "flip-esm-untrained-per-aa",
        "flip-esm1b-mean",
        "flip-esm1b-mut-mean",
        "flip-esm1b-per-aa",
        "flip-esm1v-mean",
        "flip-esm1v-mut-mean",
        "flip-esm1v-per-aa",
        "flip-levenshtein",
        "flip-ridge"
      ],
      "protocol": {
        "contamination": {
          "notes": null,
          "reporting_status": "reported",
          "value": "Mutation-depth extrapolation above two mutations."
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "deterministic regression scorer"
        },
        "reasoning": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "repeats": {
          "notes": "A common repeat count for the published table value is not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "A common random seed is not reported in the main paper.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "statistical": {
          "notes": "Mean-squared error is also evaluated in supplementary analyses and supported by the public scorer.",
          "reporting_status": "reported",
          "value": "Spearman correlation over held-out test examples; the main table reports point values without confidence intervals."
        },
        "system_prompt_public": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "temperature": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "time_budget": {
          "notes": "No common runtime budget; the paper reports Nvidia Quadro RTX A6000 hardware and omits selected infeasible per-AA runs.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "code_execution": {
            "notes": "Training code is part of the evaluation implementation, not an affordance granted to an agent.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "container": {
            "notes": "No common container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "external_tools": {
            "notes": "ESM representations are frozen; supervised heads or baselines train on the split.",
            "reporting_status": "reported",
            "value": "model-specific frozen embeddings or one-hot sequence encoding"
          },
          "internet": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          }
        },
        "turns": {
          "notes": "Batch supervised training and inference.",
          "reporting_status": "not_applicable",
          "value": "not-applicable"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-two-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1b-per-aa",
          "n": 8306,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.55
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-two-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1b-mean",
          "n": 8306,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.36
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-two-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1b-mut-mean",
          "n": 8306,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.19
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-two-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1v-per-aa",
          "n": 8306,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.28
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-two-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1v-mean",
          "n": 8306,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.32
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-two-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1v-mut-mean",
          "n": 8306,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.19
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-two-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm-untrained-per-aa",
          "n": 8306,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.06
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-two-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm-untrained-mean",
          "n": 8306,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.05
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-two-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm-untrained-mut-mean",
          "n": 8306,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.21
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-two-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-ridge",
          "n": 8306,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.59
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-two-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-cnn",
          "n": 8306,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.32
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-two-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-levenshtein",
          "n": 8306,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.16
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-gb1-two-vs-rest-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-blosum62",
          "n": 8306,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.14
        }
      ],
      "scope": {
        "filter": "Train on wild type, single, and double mutants (427 examples); test on 8,306 triple/quadruple variants.",
        "n": 8306,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "gb1-two-vs-rest-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Creator-paper split setting and all published main-table Spearman values are registered.",
        "status": "verified"
      },
      "work_id": "flip-paper",
      "work_version_id": "flip-paper-2021-10-11"
    },
    {
      "benchmark_id": "flip-meltome",
      "benchmark_version": "original-2021",
      "comparability_group": "flip-meltome-human-spearman",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "flip-meltome-human-scope-protocol",
          "locator": {
            "note": "Exact train/test counts, split rule, baseline identities, pooling, optimizer, batch sizes, early stopping, and hardware.",
            "type": "table",
            "value": "pp. 4–7, Table 2 and Sections 3–4"
          },
          "source_id": "flip-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/protocol",
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "flip-meltome-human-results",
          "locator": {
            "note": "Published held-out Spearman values; em dashes and NA entries are omitted rather than encoded as zeros.",
            "type": "table",
            "value": "p. 9, Table 6 (Human column)"
          },
          "source_id": "flip-paper",
          "source_type": "work",
          "supports": [
            "/metrics",
            "/results"
          ]
        }
      ],
      "id": "flip-meltome-human",
      "metrics": [
        {
          "aggregation": "computed across all examples in this held-out test split",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "spearman-correlation",
          "pass_threshold": null,
          "range": [
            -1,
            1
          ],
          "source_label": "Spearman correlation",
          "tolerance": null,
          "unit": "correlation"
        },
        {
          "aggregation": "mean across all examples in this held-out test split",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "mean-squared-error",
          "pass_threshold": null,
          "range": null,
          "source_label": "Mean squared error",
          "tolerance": null,
          "unit": "squared fitness units"
        }
      ],
      "model_ids": [
        "flip-cnn",
        "flip-esm-untrained-mean",
        "flip-esm-untrained-per-aa",
        "flip-esm1b-mean",
        "flip-esm1b-per-aa",
        "flip-esm1v-mean",
        "flip-esm1v-per-aa",
        "flip-ridge"
      ],
      "protocol": {
        "contamination": {
          "notes": null,
          "reporting_status": "reported",
          "value": "Human-only sequence-cluster holdout at 20% identity."
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "deterministic regression scorer"
        },
        "reasoning": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "repeats": {
          "notes": "A common repeat count for the published table value is not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "A common random seed is not reported in the main paper.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "statistical": {
          "notes": "Mean-squared error is also evaluated in supplementary analyses and supported by the public scorer.",
          "reporting_status": "reported",
          "value": "Spearman correlation over held-out test examples; the main table reports point values without confidence intervals."
        },
        "system_prompt_public": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "temperature": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "time_budget": {
          "notes": "No common runtime budget; the paper reports Nvidia Quadro RTX A6000 hardware and omits selected infeasible per-AA runs.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "code_execution": {
            "notes": "Training code is part of the evaluation implementation, not an affordance granted to an agent.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "container": {
            "notes": "No common container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "external_tools": {
            "notes": "ESM representations are frozen; supervised heads or baselines train on the split.",
            "reporting_status": "reported",
            "value": "model-specific frozen embeddings or one-hot sequence encoding"
          },
          "internet": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          }
        },
        "turns": {
          "notes": "Batch supervised training and inference.",
          "reporting_status": "not_applicable",
          "value": "not-applicable"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-meltome-human-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1b-per-aa",
          "n": 1945,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.71
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-meltome-human-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1b-mean",
          "n": 1945,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.7
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-meltome-human-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1v-per-aa",
          "n": 1945,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.77
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-meltome-human-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1v-mean",
          "n": 1945,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.75
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-meltome-human-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm-untrained-per-aa",
          "n": 1945,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.44
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-meltome-human-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm-untrained-mean",
          "n": 1945,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.48
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-meltome-human-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-ridge",
          "n": 1945,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.15
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-meltome-human-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-cnn",
          "n": 1945,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.5
        }
      ],
      "scope": {
        "filter": "Human-only 20%-identity cluster split: 8,148 training examples and 1,945 held-out cluster representatives.",
        "n": 1945,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "meltome-human-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Creator-paper split setting and all published main-table Spearman values are registered.",
        "status": "verified"
      },
      "work_id": "flip-paper",
      "work_version_id": "flip-paper-2021-10-11"
    },
    {
      "benchmark_id": "flip-meltome",
      "benchmark_version": "original-2021",
      "comparability_group": "flip-meltome-human-cell-spearman",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "flip-meltome-human-cell-scope-protocol",
          "locator": {
            "note": "Exact train/test counts, split rule, baseline identities, pooling, optimizer, batch sizes, early stopping, and hardware.",
            "type": "table",
            "value": "pp. 4–7, Table 2 and Sections 3–4"
          },
          "source_id": "flip-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/protocol",
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "flip-meltome-human-cell-results",
          "locator": {
            "note": "Published held-out Spearman values; em dashes and NA entries are omitted rather than encoded as zeros.",
            "type": "table",
            "value": "p. 9, Table 6 (Human-Cell column)"
          },
          "source_id": "flip-paper",
          "source_type": "work",
          "supports": [
            "/metrics",
            "/results"
          ]
        }
      ],
      "id": "flip-meltome-human-cell",
      "metrics": [
        {
          "aggregation": "computed across all examples in this held-out test split",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "spearman-correlation",
          "pass_threshold": null,
          "range": [
            -1,
            1
          ],
          "source_label": "Spearman correlation",
          "tolerance": null,
          "unit": "correlation"
        },
        {
          "aggregation": "mean across all examples in this held-out test split",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "mean-squared-error",
          "pass_threshold": null,
          "range": null,
          "source_label": "Mean squared error",
          "tolerance": null,
          "unit": "squared fitness units"
        }
      ],
      "model_ids": [
        "flip-cnn",
        "flip-esm-untrained-mean",
        "flip-esm-untrained-per-aa",
        "flip-esm1b-mean",
        "flip-esm1b-per-aa",
        "flip-esm1v-mean",
        "flip-esm1v-per-aa",
        "flip-ridge"
      ],
      "protocol": {
        "contamination": {
          "notes": null,
          "reporting_status": "reported",
          "value": "Single-cell-line sequence-cluster holdout at 20% identity."
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "deterministic regression scorer"
        },
        "reasoning": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "repeats": {
          "notes": "A common repeat count for the published table value is not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "A common random seed is not reported in the main paper.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "statistical": {
          "notes": "Mean-squared error is also evaluated in supplementary analyses and supported by the public scorer.",
          "reporting_status": "reported",
          "value": "Spearman correlation over held-out test examples; the main table reports point values without confidence intervals."
        },
        "system_prompt_public": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "temperature": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "time_budget": {
          "notes": "No common runtime budget; the paper reports Nvidia Quadro RTX A6000 hardware and omits selected infeasible per-AA runs.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "code_execution": {
            "notes": "Training code is part of the evaluation implementation, not an affordance granted to an agent.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "container": {
            "notes": "No common container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "external_tools": {
            "notes": "ESM representations are frozen; supervised heads or baselines train on the split.",
            "reporting_status": "reported",
            "value": "model-specific frozen embeddings or one-hot sequence encoding"
          },
          "internet": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          }
        },
        "turns": {
          "notes": "Batch supervised training and inference.",
          "reporting_status": "not_applicable",
          "value": "not-applicable"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-meltome-human-cell-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1b-per-aa",
          "n": 1366,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.76
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-meltome-human-cell-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1b-mean",
          "n": 1366,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.75
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-meltome-human-cell-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1v-per-aa",
          "n": 1366,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.78
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-meltome-human-cell-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1v-mean",
          "n": 1366,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.74
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-meltome-human-cell-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm-untrained-per-aa",
          "n": 1366,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.46
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-meltome-human-cell-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm-untrained-mean",
          "n": 1366,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.49
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-meltome-human-cell-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-ridge",
          "n": 1366,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.24
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-meltome-human-cell-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-cnn",
          "n": 1366,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.49
        }
      ],
      "scope": {
        "filter": "Single-human-cell-line 20%-identity cluster split: 5,792 training examples and 1,366 held-out cluster representatives.",
        "n": 1366,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "meltome-human-cell-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Creator-paper split setting and all published main-table Spearman values are registered.",
        "status": "verified"
      },
      "work_id": "flip-paper",
      "work_version_id": "flip-paper-2021-10-11"
    },
    {
      "benchmark_id": "flip-meltome",
      "benchmark_version": "original-2021",
      "comparability_group": "flip-meltome-mixed-spearman",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "flip-meltome-mixed-scope-protocol",
          "locator": {
            "note": "Exact train/test counts, split rule, baseline identities, pooling, optimizer, batch sizes, early stopping, and hardware.",
            "type": "table",
            "value": "pp. 4–7, Table 2 and Sections 3–4"
          },
          "source_id": "flip-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/protocol",
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "flip-meltome-mixed-results",
          "locator": {
            "note": "Published held-out Spearman values; em dashes and NA entries are omitted rather than encoded as zeros.",
            "type": "table",
            "value": "p. 9, Table 6 (Mixed column)"
          },
          "source_id": "flip-paper",
          "source_type": "work",
          "supports": [
            "/metrics",
            "/results"
          ]
        }
      ],
      "id": "flip-meltome-mixed",
      "metrics": [
        {
          "aggregation": "computed across all examples in this held-out test split",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "spearman-correlation",
          "pass_threshold": null,
          "range": [
            -1,
            1
          ],
          "source_label": "Spearman correlation",
          "tolerance": null,
          "unit": "correlation"
        },
        {
          "aggregation": "mean across all examples in this held-out test split",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "mean-squared-error",
          "pass_threshold": null,
          "range": null,
          "source_label": "Mean squared error",
          "tolerance": null,
          "unit": "squared fitness units"
        }
      ],
      "model_ids": [
        "flip-cnn",
        "flip-esm-untrained-mean",
        "flip-esm-untrained-per-aa",
        "flip-esm1b-mean",
        "flip-esm1b-per-aa",
        "flip-esm1v-mean",
        "flip-esm1v-per-aa",
        "flip-ridge"
      ],
      "protocol": {
        "contamination": {
          "notes": null,
          "reporting_status": "reported",
          "value": "Sequence-cluster holdout at 20% identity; test uses held-out cluster representatives."
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "deterministic regression scorer"
        },
        "reasoning": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "repeats": {
          "notes": "A common repeat count for the published table value is not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "A common random seed is not reported in the main paper.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "statistical": {
          "notes": "Mean-squared error is also evaluated in supplementary analyses and supported by the public scorer.",
          "reporting_status": "reported",
          "value": "Spearman correlation over held-out test examples; the main table reports point values without confidence intervals."
        },
        "system_prompt_public": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "temperature": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "time_budget": {
          "notes": "No common runtime budget; the paper reports Nvidia Quadro RTX A6000 hardware and omits selected infeasible per-AA runs.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Supervised protein regression; no in-context examples.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "code_execution": {
            "notes": "Training code is part of the evaluation implementation, not an affordance granted to an agent.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "container": {
            "notes": "No common container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "external_tools": {
            "notes": "ESM representations are frozen; supervised heads or baselines train on the split.",
            "reporting_status": "reported",
            "value": "model-specific frozen embeddings or one-hot sequence encoding"
          },
          "internet": {
            "notes": "Supervised protein regression; no in-context examples.",
            "reporting_status": "not_applicable",
            "value": null
          }
        },
        "turns": {
          "notes": "Batch supervised training and inference.",
          "reporting_status": "not_applicable",
          "value": "not-applicable"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-meltome-mixed-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1b-per-aa",
          "n": 3134,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.68
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-meltome-mixed-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1b-mean",
          "n": 3134,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.68
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-meltome-mixed-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1v-per-aa",
          "n": 3134,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.65
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-meltome-mixed-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm1v-mean",
          "n": 3134,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.67
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-meltome-mixed-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm-untrained-per-aa",
          "n": 3134,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.44
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-meltome-mixed-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-esm-untrained-mean",
          "n": 3134,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.36
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-meltome-mixed-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-ridge",
          "n": 3134,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.17
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "flip-meltome-mixed-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "flip-cnn",
          "n": 3134,
          "notes": "Held-out test-set Spearman from the creator paper.",
          "status": "verified",
          "value": 0.34
        }
      ],
      "scope": {
        "filter": "Train on all members of 80% of 20%-identity clusters (24,817 examples); test on representatives from the remaining clusters (3,134 examples).",
        "n": 3134,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "meltome-mixed-test",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Creator-paper split setting and all published main-table Spearman values are registered.",
        "status": "verified"
      },
      "work_id": "flip-paper",
      "work_version_id": "flip-paper-2021-10-11"
    },
    {
      "benchmark_id": "genebench-pro",
      "benchmark_version": "paper-v1",
      "comparability_group": "genebench-pro-paper-v1-full-claude-high",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "genebench-pro-claude-high-protocol",
          "locator": {
            "note": "Full scope, attempts, invalid-run exclusion, Docker environment, installed tools, no internet, response schema, binary grader, and no uniform harness wall-clock budget.",
            "type": "section",
            "value": "Methods — Evaluation and grading, printed page 15"
          },
          "source_id": "genebench-pro-report",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "genebench-pro-claude-high-results",
          "locator": {
            "note": "Exact configuration labels, pass rates, 95% CI bounds, average tokens where available, valid-attempt summaries, and problem-regime shares.",
            "type": "table",
            "value": "Supplementary Table 1, printed page 21"
          },
          "source_id": "genebench-pro-report",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "genebench-pro-claude-high",
      "metrics": [
        {
          "aggregation": "unweighted mean across 129 problem-level pass rates; each problem-level rate is passing valid attempts divided by valid attempts",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "pass-rate",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Eval-level pass rate",
          "tolerance": "A problem attempt passes only when all graded fields satisfy their problem-specific exact-match rules or absolute numeric tolerances.",
          "unit": "percent"
        }
      ],
      "model_ids": [
        "claude-opus-4-8"
      ],
      "protocol": {
        "contamination": {
          "notes": "The full reported evaluation includes every release stratum.",
          "reporting_status": "reported",
          "value": "constructively simulated suite with disjoint 10 public, 50 Artificial Analysis, and 69 internal-holdout release strata"
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "problem-specific Python binary checker requiring every graded field to satisfy exact-match or absolute-tolerance constraints"
        },
        "reasoning": {
          "notes": "Effort labels are provider-native configuration labels; no claim is made that equal labels imply equal token budgets across providers.",
          "reporting_status": "reported",
          "value": "provider-reported high reasoning setting"
        },
        "repeats": {
          "notes": "5 independent attempts per model–problem pair were scheduled; invalid container, tooling, provider, or response-format attempts were excluded.",
          "reporting_status": "reported",
          "value": 5
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The report enumerates the initial instructions but does not characterize the setup using a zero/few-shot convention.",
          "reporting_status": "not_reported",
          "value": null
        },
        "statistical": {
          "notes": "Per-problem rates use the final valid attempts; no partial credit. The free-text reasoning field is collected but not graded.",
          "reporting_status": "reported",
          "value": "unweighted mean of 129 per-problem pass rates with a 95% hierarchical bootstrap CI from 20,000 resamples of problems and repeated runs"
        },
        "system_prompt_public": {
          "notes": "The task prompts for ten cases are public, but the full harness system message and 119 held-out prompts are not published.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Runs remained subject to provider and platform behavior.",
          "reporting_status": "reported",
          "value": "No additional uniform wall-clock budget imposed by the harness"
        },
        "token_budget": {
          "notes": "No uniform token budget is reported. Supplementary Table 1 reports observed average tokens only for mainline GPT-family configurations.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "The isolated workspace had no internet access.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Python and R analysis were available in the Linux environment.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "Runs used a Linux Docker container.",
            "reporting_status": "reported",
            "value": true
          },
          "databases": {
            "notes": "Agents were limited to staged local files, installed software, and model-internal knowledge.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Representative installed packages; the paper provides the longer exact list.",
            "reporting_status": "reported",
            "value": [
              "Python/R scientific stacks",
              "PLINK 2.0",
              "bedtools/tabix",
              "pysam/cyvcf2",
              "scanpy/anndata",
              "DESeq2/edgeR/limma"
            ]
          },
          "internet": {
            "notes": "Methods explicitly state that the execution environment had no internet access.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "Agents iteratively inspect files, execute analyses, and return a final JSON object.",
          "reporting_status": "reported",
          "value": "multi-turn agent-container trajectory"
        }
      },
      "results": [
        {
          "ci_high": 13.0,
          "ci_low": 5.4,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-claude-high-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "claude-opus-4-8",
          "n": 129,
          "notes": "Supplementary Table 1 configuration ClaudeOpus4.8 (high); nominal attempts per problem 5, valid-attempt mean 5.0 and range 4–5; average tokens not reported; problem regimes 0%=78.3%, 0–10%=0.0%, 10–50%=14.7%, ≥50%=7.0%.",
          "status": "verified",
          "value": 9.0
        }
      ],
      "scope": {
        "filter": "Complete formal 129-problem suite, including the 10 public, 50 Artificial Analysis, and 69 internal-holdout problems.",
        "n": 129,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "1 configuration result(s), confidence intervals, repeats, valid-attempt summaries, and environment settings were checked against Methods and Supplementary Table 1.",
        "status": "verified"
      },
      "work_id": "genebench-pro-report",
      "work_version_id": "genebench-pro-report-2026-06-30"
    },
    {
      "benchmark_id": "genebench-pro",
      "benchmark_version": "paper-v1",
      "comparability_group": "genebench-pro-paper-v1-full-claude-low",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "genebench-pro-claude-low-protocol",
          "locator": {
            "note": "Full scope, attempts, invalid-run exclusion, Docker environment, installed tools, no internet, response schema, binary grader, and no uniform harness wall-clock budget.",
            "type": "section",
            "value": "Methods — Evaluation and grading, printed page 15"
          },
          "source_id": "genebench-pro-report",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "genebench-pro-claude-low-results",
          "locator": {
            "note": "Exact configuration labels, pass rates, 95% CI bounds, average tokens where available, valid-attempt summaries, and problem-regime shares.",
            "type": "table",
            "value": "Supplementary Table 1, printed page 21"
          },
          "source_id": "genebench-pro-report",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "genebench-pro-claude-low",
      "metrics": [
        {
          "aggregation": "unweighted mean across 129 problem-level pass rates; each problem-level rate is passing valid attempts divided by valid attempts",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "pass-rate",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Eval-level pass rate",
          "tolerance": "A problem attempt passes only when all graded fields satisfy their problem-specific exact-match rules or absolute numeric tolerances.",
          "unit": "percent"
        }
      ],
      "model_ids": [
        "claude-opus-4-8"
      ],
      "protocol": {
        "contamination": {
          "notes": "The full reported evaluation includes every release stratum.",
          "reporting_status": "reported",
          "value": "constructively simulated suite with disjoint 10 public, 50 Artificial Analysis, and 69 internal-holdout release strata"
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "problem-specific Python binary checker requiring every graded field to satisfy exact-match or absolute-tolerance constraints"
        },
        "reasoning": {
          "notes": "Effort labels are provider-native configuration labels; no claim is made that equal labels imply equal token budgets across providers.",
          "reporting_status": "reported",
          "value": "provider-reported low reasoning setting"
        },
        "repeats": {
          "notes": "5 independent attempts per model–problem pair were scheduled; invalid container, tooling, provider, or response-format attempts were excluded.",
          "reporting_status": "reported",
          "value": 5
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The report enumerates the initial instructions but does not characterize the setup using a zero/few-shot convention.",
          "reporting_status": "not_reported",
          "value": null
        },
        "statistical": {
          "notes": "Per-problem rates use the final valid attempts; no partial credit. The free-text reasoning field is collected but not graded.",
          "reporting_status": "reported",
          "value": "unweighted mean of 129 per-problem pass rates with a 95% hierarchical bootstrap CI from 20,000 resamples of problems and repeated runs"
        },
        "system_prompt_public": {
          "notes": "The task prompts for ten cases are public, but the full harness system message and 119 held-out prompts are not published.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Runs remained subject to provider and platform behavior.",
          "reporting_status": "reported",
          "value": "No additional uniform wall-clock budget imposed by the harness"
        },
        "token_budget": {
          "notes": "No uniform token budget is reported. Supplementary Table 1 reports observed average tokens only for mainline GPT-family configurations.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "The isolated workspace had no internet access.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Python and R analysis were available in the Linux environment.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "Runs used a Linux Docker container.",
            "reporting_status": "reported",
            "value": true
          },
          "databases": {
            "notes": "Agents were limited to staged local files, installed software, and model-internal knowledge.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Representative installed packages; the paper provides the longer exact list.",
            "reporting_status": "reported",
            "value": [
              "Python/R scientific stacks",
              "PLINK 2.0",
              "bedtools/tabix",
              "pysam/cyvcf2",
              "scanpy/anndata",
              "DESeq2/edgeR/limma"
            ]
          },
          "internet": {
            "notes": "Methods explicitly state that the execution environment had no internet access.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "Agents iteratively inspect files, execute analyses, and return a final JSON object.",
          "reporting_status": "reported",
          "value": "multi-turn agent-container trajectory"
        }
      },
      "results": [
        {
          "ci_high": 7.2,
          "ci_low": 1.9,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-claude-low-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "claude-opus-4-8",
          "n": 129,
          "notes": "Supplementary Table 1 configuration ClaudeOpus4.8 (low); nominal attempts per problem 5, valid-attempt mean 5.0 and range 3–5; average tokens not reported; problem regimes 0%=88.4%, 0–10%=0.0%, 10–50%=10.1%, ≥50%=1.6%.",
          "status": "verified",
          "value": 4.3
        }
      ],
      "scope": {
        "filter": "Complete formal 129-problem suite, including the 10 public, 50 Artificial Analysis, and 69 internal-holdout problems.",
        "n": 129,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "1 configuration result(s), confidence intervals, repeats, valid-attempt summaries, and environment settings were checked against Methods and Supplementary Table 1.",
        "status": "verified"
      },
      "work_id": "genebench-pro-report",
      "work_version_id": "genebench-pro-report-2026-06-30"
    },
    {
      "benchmark_id": "genebench-pro",
      "benchmark_version": "paper-v1",
      "comparability_group": "genebench-pro-paper-v1-full-claude-max",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "genebench-pro-claude-max-protocol",
          "locator": {
            "note": "Full scope, attempts, invalid-run exclusion, Docker environment, installed tools, no internet, response schema, binary grader, and no uniform harness wall-clock budget.",
            "type": "section",
            "value": "Methods — Evaluation and grading, printed page 15"
          },
          "source_id": "genebench-pro-report",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "genebench-pro-claude-max-results",
          "locator": {
            "note": "Exact configuration labels, pass rates, 95% CI bounds, average tokens where available, valid-attempt summaries, and problem-regime shares.",
            "type": "table",
            "value": "Supplementary Table 1, printed page 21"
          },
          "source_id": "genebench-pro-report",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "genebench-pro-claude-max",
      "metrics": [
        {
          "aggregation": "unweighted mean across 129 problem-level pass rates; each problem-level rate is passing valid attempts divided by valid attempts",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "pass-rate",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Eval-level pass rate",
          "tolerance": "A problem attempt passes only when all graded fields satisfy their problem-specific exact-match rules or absolute numeric tolerances.",
          "unit": "percent"
        }
      ],
      "model_ids": [
        "claude-opus-4-8"
      ],
      "protocol": {
        "contamination": {
          "notes": "The full reported evaluation includes every release stratum.",
          "reporting_status": "reported",
          "value": "constructively simulated suite with disjoint 10 public, 50 Artificial Analysis, and 69 internal-holdout release strata"
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "problem-specific Python binary checker requiring every graded field to satisfy exact-match or absolute-tolerance constraints"
        },
        "reasoning": {
          "notes": "Effort labels are provider-native configuration labels; no claim is made that equal labels imply equal token budgets across providers.",
          "reporting_status": "reported",
          "value": "provider-reported max reasoning setting"
        },
        "repeats": {
          "notes": "5 independent attempts per model–problem pair were scheduled; invalid container, tooling, provider, or response-format attempts were excluded.",
          "reporting_status": "reported",
          "value": 5
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The report enumerates the initial instructions but does not characterize the setup using a zero/few-shot convention.",
          "reporting_status": "not_reported",
          "value": null
        },
        "statistical": {
          "notes": "Per-problem rates use the final valid attempts; no partial credit. The free-text reasoning field is collected but not graded.",
          "reporting_status": "reported",
          "value": "unweighted mean of 129 per-problem pass rates with a 95% hierarchical bootstrap CI from 20,000 resamples of problems and repeated runs"
        },
        "system_prompt_public": {
          "notes": "The task prompts for ten cases are public, but the full harness system message and 119 held-out prompts are not published.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Runs remained subject to provider and platform behavior.",
          "reporting_status": "reported",
          "value": "No additional uniform wall-clock budget imposed by the harness"
        },
        "token_budget": {
          "notes": "No uniform token budget is reported. Supplementary Table 1 reports observed average tokens only for mainline GPT-family configurations.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "The isolated workspace had no internet access.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Python and R analysis were available in the Linux environment.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "Runs used a Linux Docker container.",
            "reporting_status": "reported",
            "value": true
          },
          "databases": {
            "notes": "Agents were limited to staged local files, installed software, and model-internal knowledge.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Representative installed packages; the paper provides the longer exact list.",
            "reporting_status": "reported",
            "value": [
              "Python/R scientific stacks",
              "PLINK 2.0",
              "bedtools/tabix",
              "pysam/cyvcf2",
              "scanpy/anndata",
              "DESeq2/edgeR/limma"
            ]
          },
          "internet": {
            "notes": "Methods explicitly state that the execution environment had no internet access.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "Agents iteratively inspect files, execute analyses, and return a final JSON object.",
          "reporting_status": "reported",
          "value": "multi-turn agent-container trajectory"
        }
      },
      "results": [
        {
          "ci_high": 21.2,
          "ci_low": 11.0,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-claude-max-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "claude-opus-4-8",
          "n": 129,
          "notes": "Supplementary Table 1 configuration ClaudeOpus4.8 (max); nominal attempts per problem 5, valid-attempt mean 5.0 and range 5–5; average tokens not reported; problem regimes 0%=67.4%, 0–10%=0.0%, 10–50%=18.6%, ≥50%=14.0%.",
          "status": "verified",
          "value": 16.0
        }
      ],
      "scope": {
        "filter": "Complete formal 129-problem suite, including the 10 public, 50 Artificial Analysis, and 69 internal-holdout problems.",
        "n": 129,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "1 configuration result(s), confidence intervals, repeats, valid-attempt summaries, and environment settings were checked against Methods and Supplementary Table 1.",
        "status": "verified"
      },
      "work_id": "genebench-pro-report",
      "work_version_id": "genebench-pro-report-2026-06-30"
    },
    {
      "benchmark_id": "genebench-pro",
      "benchmark_version": "paper-v1",
      "comparability_group": "genebench-pro-paper-v1-full-claude-medium",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "genebench-pro-claude-medium-protocol",
          "locator": {
            "note": "Full scope, attempts, invalid-run exclusion, Docker environment, installed tools, no internet, response schema, binary grader, and no uniform harness wall-clock budget.",
            "type": "section",
            "value": "Methods — Evaluation and grading, printed page 15"
          },
          "source_id": "genebench-pro-report",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "genebench-pro-claude-medium-results",
          "locator": {
            "note": "Exact configuration labels, pass rates, 95% CI bounds, average tokens where available, valid-attempt summaries, and problem-regime shares.",
            "type": "table",
            "value": "Supplementary Table 1, printed page 21"
          },
          "source_id": "genebench-pro-report",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "genebench-pro-claude-medium",
      "metrics": [
        {
          "aggregation": "unweighted mean across 129 problem-level pass rates; each problem-level rate is passing valid attempts divided by valid attempts",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "pass-rate",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Eval-level pass rate",
          "tolerance": "A problem attempt passes only when all graded fields satisfy their problem-specific exact-match rules or absolute numeric tolerances.",
          "unit": "percent"
        }
      ],
      "model_ids": [
        "claude-opus-4-8"
      ],
      "protocol": {
        "contamination": {
          "notes": "The full reported evaluation includes every release stratum.",
          "reporting_status": "reported",
          "value": "constructively simulated suite with disjoint 10 public, 50 Artificial Analysis, and 69 internal-holdout release strata"
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "problem-specific Python binary checker requiring every graded field to satisfy exact-match or absolute-tolerance constraints"
        },
        "reasoning": {
          "notes": "Effort labels are provider-native configuration labels; no claim is made that equal labels imply equal token budgets across providers.",
          "reporting_status": "reported",
          "value": "provider-reported medium reasoning setting"
        },
        "repeats": {
          "notes": "5 independent attempts per model–problem pair were scheduled; invalid container, tooling, provider, or response-format attempts were excluded.",
          "reporting_status": "reported",
          "value": 5
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The report enumerates the initial instructions but does not characterize the setup using a zero/few-shot convention.",
          "reporting_status": "not_reported",
          "value": null
        },
        "statistical": {
          "notes": "Per-problem rates use the final valid attempts; no partial credit. The free-text reasoning field is collected but not graded.",
          "reporting_status": "reported",
          "value": "unweighted mean of 129 per-problem pass rates with a 95% hierarchical bootstrap CI from 20,000 resamples of problems and repeated runs"
        },
        "system_prompt_public": {
          "notes": "The task prompts for ten cases are public, but the full harness system message and 119 held-out prompts are not published.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Runs remained subject to provider and platform behavior.",
          "reporting_status": "reported",
          "value": "No additional uniform wall-clock budget imposed by the harness"
        },
        "token_budget": {
          "notes": "No uniform token budget is reported. Supplementary Table 1 reports observed average tokens only for mainline GPT-family configurations.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "The isolated workspace had no internet access.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Python and R analysis were available in the Linux environment.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "Runs used a Linux Docker container.",
            "reporting_status": "reported",
            "value": true
          },
          "databases": {
            "notes": "Agents were limited to staged local files, installed software, and model-internal knowledge.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Representative installed packages; the paper provides the longer exact list.",
            "reporting_status": "reported",
            "value": [
              "Python/R scientific stacks",
              "PLINK 2.0",
              "bedtools/tabix",
              "pysam/cyvcf2",
              "scanpy/anndata",
              "DESeq2/edgeR/limma"
            ]
          },
          "internet": {
            "notes": "Methods explicitly state that the execution environment had no internet access.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "Agents iteratively inspect files, execute analyses, and return a final JSON object.",
          "reporting_status": "reported",
          "value": "multi-turn agent-container trajectory"
        }
      },
      "results": [
        {
          "ci_high": 7.3,
          "ci_low": 2.0,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-claude-medium-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "claude-opus-4-8",
          "n": 129,
          "notes": "Supplementary Table 1 configuration ClaudeOpus4.8 (medium); nominal attempts per problem 5, valid-attempt mean 5.0 and range 5–5; average tokens not reported; problem regimes 0%=87.6%, 0–10%=0.0%, 10–50%=10.1%, ≥50%=2.3%.",
          "status": "verified",
          "value": 4.3
        }
      ],
      "scope": {
        "filter": "Complete formal 129-problem suite, including the 10 public, 50 Artificial Analysis, and 69 internal-holdout problems.",
        "n": 129,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "1 configuration result(s), confidence intervals, repeats, valid-attempt summaries, and environment settings were checked against Methods and Supplementary Table 1.",
        "status": "verified"
      },
      "work_id": "genebench-pro-report",
      "work_version_id": "genebench-pro-report-2026-06-30"
    },
    {
      "benchmark_id": "genebench-pro",
      "benchmark_version": "paper-v1",
      "comparability_group": "genebench-pro-paper-v1-full-claude-xhigh",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "genebench-pro-claude-xhigh-protocol",
          "locator": {
            "note": "Full scope, attempts, invalid-run exclusion, Docker environment, installed tools, no internet, response schema, binary grader, and no uniform harness wall-clock budget.",
            "type": "section",
            "value": "Methods — Evaluation and grading, printed page 15"
          },
          "source_id": "genebench-pro-report",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "genebench-pro-claude-xhigh-results",
          "locator": {
            "note": "Exact configuration labels, pass rates, 95% CI bounds, average tokens where available, valid-attempt summaries, and problem-regime shares.",
            "type": "table",
            "value": "Supplementary Table 1, printed page 21"
          },
          "source_id": "genebench-pro-report",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "genebench-pro-claude-xhigh",
      "metrics": [
        {
          "aggregation": "unweighted mean across 129 problem-level pass rates; each problem-level rate is passing valid attempts divided by valid attempts",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "pass-rate",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Eval-level pass rate",
          "tolerance": "A problem attempt passes only when all graded fields satisfy their problem-specific exact-match rules or absolute numeric tolerances.",
          "unit": "percent"
        }
      ],
      "model_ids": [
        "claude-opus-4-8"
      ],
      "protocol": {
        "contamination": {
          "notes": "The full reported evaluation includes every release stratum.",
          "reporting_status": "reported",
          "value": "constructively simulated suite with disjoint 10 public, 50 Artificial Analysis, and 69 internal-holdout release strata"
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "problem-specific Python binary checker requiring every graded field to satisfy exact-match or absolute-tolerance constraints"
        },
        "reasoning": {
          "notes": "Effort labels are provider-native configuration labels; no claim is made that equal labels imply equal token budgets across providers.",
          "reporting_status": "reported",
          "value": "provider-reported xhigh reasoning setting"
        },
        "repeats": {
          "notes": "5 independent attempts per model–problem pair were scheduled; invalid container, tooling, provider, or response-format attempts were excluded.",
          "reporting_status": "reported",
          "value": 5
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The report enumerates the initial instructions but does not characterize the setup using a zero/few-shot convention.",
          "reporting_status": "not_reported",
          "value": null
        },
        "statistical": {
          "notes": "Per-problem rates use the final valid attempts; no partial credit. The free-text reasoning field is collected but not graded.",
          "reporting_status": "reported",
          "value": "unweighted mean of 129 per-problem pass rates with a 95% hierarchical bootstrap CI from 20,000 resamples of problems and repeated runs"
        },
        "system_prompt_public": {
          "notes": "The task prompts for ten cases are public, but the full harness system message and 119 held-out prompts are not published.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Runs remained subject to provider and platform behavior.",
          "reporting_status": "reported",
          "value": "No additional uniform wall-clock budget imposed by the harness"
        },
        "token_budget": {
          "notes": "No uniform token budget is reported. Supplementary Table 1 reports observed average tokens only for mainline GPT-family configurations.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "The isolated workspace had no internet access.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Python and R analysis were available in the Linux environment.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "Runs used a Linux Docker container.",
            "reporting_status": "reported",
            "value": true
          },
          "databases": {
            "notes": "Agents were limited to staged local files, installed software, and model-internal knowledge.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Representative installed packages; the paper provides the longer exact list.",
            "reporting_status": "reported",
            "value": [
              "Python/R scientific stacks",
              "PLINK 2.0",
              "bedtools/tabix",
              "pysam/cyvcf2",
              "scanpy/anndata",
              "DESeq2/edgeR/limma"
            ]
          },
          "internet": {
            "notes": "Methods explicitly state that the execution environment had no internet access.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "Agents iteratively inspect files, execute analyses, and return a final JSON object.",
          "reporting_status": "reported",
          "value": "multi-turn agent-container trajectory"
        }
      },
      "results": [
        {
          "ci_high": 14.3,
          "ci_low": 6.4,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-claude-xhigh-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "claude-opus-4-8",
          "n": 129,
          "notes": "Supplementary Table 1 configuration ClaudeOpus4.8 (xhigh); nominal attempts per problem 5, valid-attempt mean 5.0 and range 4–5; average tokens not reported; problem regimes 0%=75.2%, 0–10%=0.0%, 10–50%=17.1%, ≥50%=7.8%.",
          "status": "verified",
          "value": 10.1
        }
      ],
      "scope": {
        "filter": "Complete formal 129-problem suite, including the 10 public, 50 Artificial Analysis, and 69 internal-holdout problems.",
        "n": 129,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "1 configuration result(s), confidence intervals, repeats, valid-attempt summaries, and environment settings were checked against Methods and Supplementary Table 1.",
        "status": "verified"
      },
      "work_id": "genebench-pro-report",
      "work_version_id": "genebench-pro-report-2026-06-30"
    },
    {
      "benchmark_id": "genebench-pro",
      "benchmark_version": "paper-v1",
      "comparability_group": "genebench-pro-paper-v1-full-standard-xhigh",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "genebench-pro-official-protocol",
          "locator": {
            "note": "Full scope, attempts, invalid-run exclusion, Docker environment, installed tools, no internet, response schema, binary grader, and no uniform harness wall-clock budget.",
            "type": "section",
            "value": "Methods — Evaluation and grading, printed page 15"
          },
          "source_id": "genebench-pro-report",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "genebench-pro-official-results",
          "locator": {
            "note": "Exact configuration labels, pass rates, 95% CI bounds, average tokens where available, valid-attempt summaries, and problem-regime shares.",
            "type": "table",
            "value": "Supplementary Table 1, printed page 21"
          },
          "source_id": "genebench-pro-report",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "genebench-pro-official",
      "metrics": [
        {
          "aggregation": "unweighted mean across 129 problem-level pass rates; each problem-level rate is passing valid attempts divided by valid attempts",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "pass-rate",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Eval-level pass rate",
          "tolerance": "A problem attempt passes only when all graded fields satisfy their problem-specific exact-match rules or absolute numeric tolerances.",
          "unit": "percent"
        }
      ],
      "model_ids": [
        "deepseek-v4-flash",
        "deepseek-v4-pro",
        "glm-5-1",
        "gpt-5-2",
        "gpt-5-4",
        "gpt-5-5",
        "gpt-5-6-luna",
        "gpt-5-6-sol",
        "gpt-5-6-terra",
        "kimi-k2-6",
        "mimo-v2-5",
        "mimo-v2-5-pro",
        "minimax-m2-7",
        "minimax-m3",
        "qwen-3-7-max",
        "qwen-3-7-plus",
        "tencent-hy-3-preview"
      ],
      "protocol": {
        "contamination": {
          "notes": "The full reported evaluation includes every release stratum.",
          "reporting_status": "reported",
          "value": "constructively simulated suite with disjoint 10 public, 50 Artificial Analysis, and 69 internal-holdout release strata"
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "problem-specific Python binary checker requiring every graded field to satisfy exact-match or absolute-tolerance constraints"
        },
        "reasoning": {
          "notes": "Effort labels are provider-native configuration labels; no claim is made that equal labels imply equal token budgets across providers.",
          "reporting_status": "reported",
          "value": "provider-reported xhigh reasoning setting"
        },
        "repeats": {
          "notes": "10 independent attempts per model–problem pair were scheduled; invalid container, tooling, provider, or response-format attempts were excluded.",
          "reporting_status": "reported",
          "value": 10
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The report enumerates the initial instructions but does not characterize the setup using a zero/few-shot convention.",
          "reporting_status": "not_reported",
          "value": null
        },
        "statistical": {
          "notes": "Per-problem rates use the final valid attempts; no partial credit. The free-text reasoning field is collected but not graded.",
          "reporting_status": "reported",
          "value": "unweighted mean of 129 per-problem pass rates with a 95% hierarchical bootstrap CI from 20,000 resamples of problems and repeated runs"
        },
        "system_prompt_public": {
          "notes": "The task prompts for ten cases are public, but the full harness system message and 119 held-out prompts are not published.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Runs remained subject to provider and platform behavior.",
          "reporting_status": "reported",
          "value": "No additional uniform wall-clock budget imposed by the harness"
        },
        "token_budget": {
          "notes": "No uniform token budget is reported. Supplementary Table 1 reports observed average tokens only for mainline GPT-family configurations.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "The isolated workspace had no internet access.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Python and R analysis were available in the Linux environment.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "Runs used a Linux Docker container.",
            "reporting_status": "reported",
            "value": true
          },
          "databases": {
            "notes": "Agents were limited to staged local files, installed software, and model-internal knowledge.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Representative installed packages; the paper provides the longer exact list.",
            "reporting_status": "reported",
            "value": [
              "Python/R scientific stacks",
              "PLINK 2.0",
              "bedtools/tabix",
              "pysam/cyvcf2",
              "scanpy/anndata",
              "DESeq2/edgeR/limma"
            ]
          },
          "internet": {
            "notes": "Methods explicitly state that the execution environment had no internet access.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "Agents iteratively inspect files, execute analyses, and return a final JSON object.",
          "reporting_status": "reported",
          "value": "multi-turn agent-container trajectory"
        }
      },
      "results": [
        {
          "ci_high": 1.4,
          "ci_low": 0.1,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-official-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "minimax-m2-7",
          "n": 129,
          "notes": "Supplementary Table 1 configuration MiniMaxM2.7 (xhigh); nominal attempts per problem 10, valid-attempt mean 9.9 and range 8–10; average tokens not reported; problem regimes 0%=96.1%, 0–10%=2.3%, 10–50%=1.6%, ≥50%=0.0%.",
          "status": "verified",
          "value": 0.6
        },
        {
          "ci_high": 1.7,
          "ci_low": 0.2,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-official-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "tencent-hy-3-preview",
          "n": 129,
          "notes": "Supplementary Table 1 configuration TencentHY3Preview (xhigh); nominal attempts per problem 10, valid-attempt mean 9.8 and range 8–10; average tokens not reported; problem regimes 0%=92.2%, 0–10%=6.2%, 10–50%=1.6%, ≥50%=0.0%.",
          "status": "verified",
          "value": 0.9
        },
        {
          "ci_high": 2.0,
          "ci_low": 0.1,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-official-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "minimax-m3",
          "n": 129,
          "notes": "Supplementary Table 1 configuration MiniMaxM3 (xhigh); nominal attempts per problem 10, valid-attempt mean 9.6 and range 7–10; average tokens not reported; problem regimes 0%=96.1%, 0–10%=1.6%, 10–50%=2.3%, ≥50%=0.0%.",
          "status": "verified",
          "value": 0.9
        },
        {
          "ci_high": 2.2,
          "ci_low": 0.4,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-official-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "mimo-v2-5",
          "n": 129,
          "notes": "Supplementary Table 1 configuration MiMoV2.5 (xhigh); nominal attempts per problem 10, valid-attempt mean 9.9 and range 8–10; average tokens not reported; problem regimes 0%=91.5%, 0–10%=6.2%, 10–50%=2.3%, ≥50%=0.0%.",
          "status": "verified",
          "value": 1.2
        },
        {
          "ci_high": 2.8,
          "ci_low": 0.2,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-official-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "glm-5-1",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GLM5.1 (xhigh); nominal attempts per problem 10, valid-attempt mean 9.7 and range 8–10; average tokens not reported; problem regimes 0%=95.3%, 0–10%=2.3%, 10–50%=1.6%, ≥50%=0.8%.",
          "status": "verified",
          "value": 1.2
        },
        {
          "ci_high": 3.5,
          "ci_low": 0.9,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-official-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "mimo-v2-5-pro",
          "n": 129,
          "notes": "Supplementary Table 1 configuration MiMoV2.5Pro (xhigh); nominal attempts per problem 10, valid-attempt mean 9.9 and range 8–10; average tokens not reported; problem regimes 0%=86.8%, 0–10%=9.3%, 10–50%=3.1%, ≥50%=0.8%.",
          "status": "verified",
          "value": 2.0
        },
        {
          "ci_high": 4.0,
          "ci_low": 1.0,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-official-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "qwen-3-7-plus",
          "n": 129,
          "notes": "Supplementary Table 1 configuration Qwen3.7Plus (xhigh); nominal attempts per problem 10, valid-attempt mean 9.7 and range 6–10; average tokens not reported; problem regimes 0%=86.8%, 0–10%=7.0%, 10–50%=5.4%, ≥50%=0.8%.",
          "status": "verified",
          "value": 2.3
        },
        {
          "ci_high": 4.2,
          "ci_low": 1.0,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-official-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "deepseek-v4-flash",
          "n": 129,
          "notes": "Supplementary Table 1 configuration DeepSeekV4Flash (xhigh); nominal attempts per problem 10, valid-attempt mean 9.9 and range 9–10; average tokens not reported; problem regimes 0%=87.6%, 0–10%=6.2%, 10–50%=4.7%, ≥50%=1.6%.",
          "status": "verified",
          "value": 2.4
        },
        {
          "ci_high": 3.9,
          "ci_low": 1.2,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-official-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "deepseek-v4-pro",
          "n": 129,
          "notes": "Supplementary Table 1 configuration DeepSeekV4Pro (xhigh); nominal attempts per problem 10, valid-attempt mean 9.6 and range 8–10; average tokens not reported; problem regimes 0%=84.5%, 0–10%=4.7%, 10–50%=10.9%, ≥50%=0.0%.",
          "status": "verified",
          "value": 2.4
        },
        {
          "ci_high": 6.0,
          "ci_low": 2.2,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-official-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "qwen-3-7-max",
          "n": 129,
          "notes": "Supplementary Table 1 configuration Qwen3.7Max (xhigh); nominal attempts per problem 10, valid-attempt mean 9.8 and range 9–10; average tokens not reported; problem regimes 0%=79.8%, 0–10%=7.8%, 10–50%=11.6%, ≥50%=0.8%.",
          "status": "verified",
          "value": 4.0
        },
        {
          "ci_high": 7.2,
          "ci_low": 2.0,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-official-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "kimi-k2-6",
          "n": 129,
          "notes": "Supplementary Table 1 configuration KimiK2.6 (xhigh); nominal attempts per problem 10, valid-attempt mean 9.9 and range 7–10; average tokens not reported; problem regimes 0%=84.5%, 0–10%=6.2%, 10–50%=6.2%, ≥50%=3.1%.",
          "status": "verified",
          "value": 4.4
        },
        {
          "ci_high": 7.4,
          "ci_low": 2.7,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-official-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-2",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.2 (xhigh); nominal attempts per problem 10, valid-attempt mean 9.9 and range 9–10; average tokens 52.0k; problem regimes 0%=77.5%, 0–10%=10.1%, 10–50%=10.9%, ≥50%=1.6%.",
          "status": "verified",
          "value": 4.9
        },
        {
          "ci_high": 12.1,
          "ci_low": 6.0,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-official-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-4",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.4 (xhigh); nominal attempts per problem 10, valid-attempt mean 10.0 and range 9–10; average tokens 44.7k; problem regimes 0%=67.4%, 0–10%=9.3%, 10–50%=18.6%, ≥50%=4.7%.",
          "status": "verified",
          "value": 8.9
        },
        {
          "ci_high": 16.1,
          "ci_low": 8.2,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-official-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-5",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.5 (xhigh); nominal attempts per problem 10, valid-attempt mean 10.0 and range 9–10; average tokens 28.7k; problem regimes 0%=64.3%, 0–10%=8.5%, 10–50%=18.6%, ≥50%=8.5%.",
          "status": "verified",
          "value": 12.0
        },
        {
          "ci_high": 15.0,
          "ci_low": 7.0,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-official-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-6-luna",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.6Luna (xhigh); nominal attempts per problem 10, valid-attempt mean 9.8 and range 7–10; average tokens 53.1k; problem regimes 0%=70.5%, 0–10%=8.5%, 10–50%=10.1%, ≥50%=10.9%.",
          "status": "verified",
          "value": 10.8
        },
        {
          "ci_high": 24.3,
          "ci_low": 13.7,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-official-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-6-terra",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.6Terra (xhigh); nominal attempts per problem 10, valid-attempt mean 10.0 and range 9–10; average tokens 31.1k; problem regimes 0%=56.6%, 0–10%=7.8%, 10–50%=19.4%, ≥50%=16.3%.",
          "status": "verified",
          "value": 18.8
        },
        {
          "ci_high": 33.2,
          "ci_low": 20.7,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-official-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-6-sol",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.6Sol (xhigh); nominal attempts per problem 10, valid-attempt mean 9.9 and range 9–10; average tokens 25.7k; problem regimes 0%=50.4%, 0–10%=6.2%, 10–50%=14.0%, ≥50%=29.5%.",
          "status": "verified",
          "value": 26.8
        }
      ],
      "scope": {
        "filter": "Complete formal 129-problem suite, including the 10 public, 50 Artificial Analysis, and 69 internal-holdout problems.",
        "n": 129,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "17 configuration result(s), confidence intervals, repeats, valid-attempt summaries, and environment settings were checked against Methods and Supplementary Table 1.",
        "status": "verified"
      },
      "work_id": "genebench-pro-report",
      "work_version_id": "genebench-pro-report-2026-06-30"
    },
    {
      "benchmark_id": "genebench-pro",
      "benchmark_version": "paper-v1",
      "comparability_group": "genebench-pro-paper-v1-full-pro-extended",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "genebench-pro-pro-mode-protocol",
          "locator": {
            "note": "Full scope, attempts, invalid-run exclusion, Docker environment, installed tools, no internet, response schema, binary grader, and no uniform harness wall-clock budget.",
            "type": "section",
            "value": "Methods — Evaluation and grading, printed page 15"
          },
          "source_id": "genebench-pro-report",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "genebench-pro-pro-mode-results",
          "locator": {
            "note": "Exact configuration labels, pass rates, 95% CI bounds, average tokens where available, valid-attempt summaries, and problem-regime shares.",
            "type": "table",
            "value": "Supplementary Table 1, printed page 21"
          },
          "source_id": "genebench-pro-report",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "genebench-pro-pro-mode",
      "metrics": [
        {
          "aggregation": "unweighted mean across 129 problem-level pass rates; each problem-level rate is passing valid attempts divided by valid attempts",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "pass-rate",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Eval-level pass rate",
          "tolerance": "A problem attempt passes only when all graded fields satisfy their problem-specific exact-match rules or absolute numeric tolerances.",
          "unit": "percent"
        }
      ],
      "model_ids": [
        "gpt-5-2-pro",
        "gpt-5-4-pro",
        "gpt-5-5-pro",
        "gpt-5-6-luna-pro",
        "gpt-5-6-pro",
        "gpt-5-6-terra-pro"
      ],
      "protocol": {
        "contamination": {
          "notes": "The full reported evaluation includes every release stratum.",
          "reporting_status": "reported",
          "value": "constructively simulated suite with disjoint 10 public, 50 Artificial Analysis, and 69 internal-holdout release strata"
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "problem-specific Python binary checker requiring every graded field to satisfy exact-match or absolute-tolerance constraints"
        },
        "reasoning": {
          "notes": "Effort labels are provider-native configuration labels; no claim is made that equal labels imply equal token budgets across providers.",
          "reporting_status": "reported",
          "value": "GPT Pro (Extended)"
        },
        "repeats": {
          "notes": "5 independent attempts per model–problem pair were scheduled; invalid container, tooling, provider, or response-format attempts were excluded.",
          "reporting_status": "reported",
          "value": 5
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The report enumerates the initial instructions but does not characterize the setup using a zero/few-shot convention.",
          "reporting_status": "not_reported",
          "value": null
        },
        "statistical": {
          "notes": "Per-problem rates use the final valid attempts; no partial credit. The free-text reasoning field is collected but not graded.",
          "reporting_status": "reported",
          "value": "unweighted mean of 129 per-problem pass rates with a 95% hierarchical bootstrap CI from 20,000 resamples of problems and repeated runs"
        },
        "system_prompt_public": {
          "notes": "The task prompts for ten cases are public, but the full harness system message and 119 held-out prompts are not published.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Runs remained subject to provider and platform behavior.",
          "reporting_status": "reported",
          "value": "No additional uniform wall-clock budget imposed by the harness"
        },
        "token_budget": {
          "notes": "No uniform token budget is reported. Supplementary Table 1 reports observed average tokens only for mainline GPT-family configurations.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "The isolated workspace had no internet access.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Python and R analysis were available in the Linux environment.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "Runs used a Linux Docker container.",
            "reporting_status": "reported",
            "value": true
          },
          "databases": {
            "notes": "Agents were limited to staged local files, installed software, and model-internal knowledge.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Representative installed packages; the paper provides the longer exact list.",
            "reporting_status": "reported",
            "value": [
              "Python/R scientific stacks",
              "PLINK 2.0",
              "bedtools/tabix",
              "pysam/cyvcf2",
              "scanpy/anndata",
              "DESeq2/edgeR/limma"
            ]
          },
          "internet": {
            "notes": "Methods explicitly state that the execution environment had no internet access.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "Agents iteratively inspect files, execute analyses, and return a final JSON object.",
          "reporting_status": "reported",
          "value": "multi-turn agent-container trajectory"
        }
      },
      "results": [
        {
          "ci_high": 12.6,
          "ci_low": 5.0,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-pro-mode-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-2-pro",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.2Pro (Extended); nominal attempts per problem 5, valid-attempt mean 5.0 and range 4–5; average tokens not reported; problem regimes 0%=79.1%, 0–10%=0.0%, 10–50%=14.0%, ≥50%=7.0%.",
          "status": "verified",
          "value": 8.5
        },
        {
          "ci_high": 22.0,
          "ci_low": 10.9,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-pro-mode-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-4-pro",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.4Pro (Extended); nominal attempts per problem 5, valid-attempt mean 5.0 and range 5–5; average tokens not reported; problem regimes 0%=72.1%, 0–10%=0.0%, 10–50%=12.4%, ≥50%=15.5%.",
          "status": "verified",
          "value": 16.3
        },
        {
          "ci_high": 26.5,
          "ci_low": 14.7,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-pro-mode-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-5-pro",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.5Pro (Extended); nominal attempts per problem 5, valid-attempt mean 5.0 and range 5–5; average tokens not reported; problem regimes 0%=66.7%, 0–10%=0.0%, 10–50%=14.0%, ≥50%=19.4%.",
          "status": "verified",
          "value": 20.5
        },
        {
          "ci_high": 30.2,
          "ci_low": 17.2,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-pro-mode-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-6-luna-pro",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.6LunaPro (Extended); nominal attempts per problem 5, valid-attempt mean 5.0 and range 5–5; average tokens not reported; problem regimes 0%=65.9%, 0–10%=0.0%, 10–50%=10.9%, ≥50%=23.3%.",
          "status": "verified",
          "value": 23.6
        },
        {
          "ci_high": 35.7,
          "ci_low": 21.7,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-pro-mode-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-6-terra-pro",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.6TerraPro (Extended); nominal attempts per problem 5, valid-attempt mean 5.0 and range 5–5; average tokens not reported; problem regimes 0%=59.7%, 0–10%=0.0%, 10–50%=11.6%, ≥50%=28.7%.",
          "status": "verified",
          "value": 28.5
        },
        {
          "ci_high": 38.9,
          "ci_low": 24.3,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-pro-mode-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-6-pro",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.6SolPro (Extended); nominal attempts per problem 5, valid-attempt mean 5.0 and range 5–5; average tokens not reported; problem regimes 0%=55.8%, 0–10%=0.0%, 10–50%=14.0%, ≥50%=30.2%.",
          "status": "verified",
          "value": 31.5
        }
      ],
      "scope": {
        "filter": "Complete formal 129-problem suite, including the 10 public, 50 Artificial Analysis, and 69 internal-holdout problems.",
        "n": 129,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "6 configuration result(s), confidence intervals, repeats, valid-attempt summaries, and environment settings were checked against Methods and Supplementary Table 1.",
        "status": "verified"
      },
      "work_id": "genebench-pro-report",
      "work_version_id": "genebench-pro-report-2026-06-30"
    },
    {
      "benchmark_id": "genebench-pro",
      "benchmark_version": "paper-v1",
      "comparability_group": "genebench-pro-paper-v1-full-reasoning-enabled",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "genebench-pro-reasoning-enabled-protocol",
          "locator": {
            "note": "Full scope, attempts, invalid-run exclusion, Docker environment, installed tools, no internet, response schema, binary grader, and no uniform harness wall-clock budget.",
            "type": "section",
            "value": "Methods — Evaluation and grading, printed page 15"
          },
          "source_id": "genebench-pro-report",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "genebench-pro-reasoning-enabled-results",
          "locator": {
            "note": "Exact configuration labels, pass rates, 95% CI bounds, average tokens where available, valid-attempt summaries, and problem-regime shares.",
            "type": "table",
            "value": "Supplementary Table 1, printed page 21"
          },
          "source_id": "genebench-pro-report",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "genebench-pro-reasoning-enabled",
      "metrics": [
        {
          "aggregation": "unweighted mean across 129 problem-level pass rates; each problem-level rate is passing valid attempts divided by valid attempts",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "pass-rate",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Eval-level pass rate",
          "tolerance": "A problem attempt passes only when all graded fields satisfy their problem-specific exact-match rules or absolute numeric tolerances.",
          "unit": "percent"
        }
      ],
      "model_ids": [
        "kimi-k2-7-code"
      ],
      "protocol": {
        "contamination": {
          "notes": "The full reported evaluation includes every release stratum.",
          "reporting_status": "reported",
          "value": "constructively simulated suite with disjoint 10 public, 50 Artificial Analysis, and 69 internal-holdout release strata"
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "problem-specific Python binary checker requiring every graded field to satisfy exact-match or absolute-tolerance constraints"
        },
        "reasoning": {
          "notes": "Effort labels are provider-native configuration labels; no claim is made that equal labels imply equal token budgets across providers.",
          "reporting_status": "reported",
          "value": "reasoning_enabled"
        },
        "repeats": {
          "notes": "10 independent attempts per model–problem pair were scheduled; invalid container, tooling, provider, or response-format attempts were excluded.",
          "reporting_status": "reported",
          "value": 10
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The report enumerates the initial instructions but does not characterize the setup using a zero/few-shot convention.",
          "reporting_status": "not_reported",
          "value": null
        },
        "statistical": {
          "notes": "Per-problem rates use the final valid attempts; no partial credit. The free-text reasoning field is collected but not graded.",
          "reporting_status": "reported",
          "value": "unweighted mean of 129 per-problem pass rates with a 95% hierarchical bootstrap CI from 20,000 resamples of problems and repeated runs"
        },
        "system_prompt_public": {
          "notes": "The task prompts for ten cases are public, but the full harness system message and 119 held-out prompts are not published.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Runs remained subject to provider and platform behavior.",
          "reporting_status": "reported",
          "value": "No additional uniform wall-clock budget imposed by the harness"
        },
        "token_budget": {
          "notes": "No uniform token budget is reported. Supplementary Table 1 reports observed average tokens only for mainline GPT-family configurations.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "The isolated workspace had no internet access.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Python and R analysis were available in the Linux environment.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "Runs used a Linux Docker container.",
            "reporting_status": "reported",
            "value": true
          },
          "databases": {
            "notes": "Agents were limited to staged local files, installed software, and model-internal knowledge.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Representative installed packages; the paper provides the longer exact list.",
            "reporting_status": "reported",
            "value": [
              "Python/R scientific stacks",
              "PLINK 2.0",
              "bedtools/tabix",
              "pysam/cyvcf2",
              "scanpy/anndata",
              "DESeq2/edgeR/limma"
            ]
          },
          "internet": {
            "notes": "Methods explicitly state that the execution environment had no internet access.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "Agents iteratively inspect files, execute analyses, and return a final JSON object.",
          "reporting_status": "reported",
          "value": "multi-turn agent-container trajectory"
        }
      },
      "results": [
        {
          "ci_high": 3.7,
          "ci_low": 1.1,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-reasoning-enabled-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "kimi-k2-7-code",
          "n": 129,
          "notes": "Supplementary Table 1 configuration KimiK2.7Code (reasoning_enabled); nominal attempts per problem 10, valid-attempt mean 9.9 and range 9–10; average tokens not reported; problem regimes 0%=84.5%, 0–10%=9.3%, 10–50%=6.2%, ≥50%=0.0%.",
          "status": "verified",
          "value": 2.3
        }
      ],
      "scope": {
        "filter": "Complete formal 129-problem suite, including the 10 public, 50 Artificial Analysis, and 69 internal-holdout problems.",
        "n": 129,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "1 configuration result(s), confidence intervals, repeats, valid-attempt summaries, and environment settings were checked against Methods and Supplementary Table 1.",
        "status": "verified"
      },
      "work_id": "genebench-pro-report",
      "work_version_id": "genebench-pro-report-2026-06-30"
    },
    {
      "benchmark_id": "genebench-pro",
      "benchmark_version": "paper-v1",
      "comparability_group": "genebench-pro-paper-v1-full-standard-high",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "genebench-pro-standard-high-protocol",
          "locator": {
            "note": "Full scope, attempts, invalid-run exclusion, Docker environment, installed tools, no internet, response schema, binary grader, and no uniform harness wall-clock budget.",
            "type": "section",
            "value": "Methods — Evaluation and grading, printed page 15"
          },
          "source_id": "genebench-pro-report",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "genebench-pro-standard-high-results",
          "locator": {
            "note": "Exact configuration labels, pass rates, 95% CI bounds, average tokens where available, valid-attempt summaries, and problem-regime shares.",
            "type": "table",
            "value": "Supplementary Table 1, printed page 21"
          },
          "source_id": "genebench-pro-report",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "genebench-pro-standard-high",
      "metrics": [
        {
          "aggregation": "unweighted mean across 129 problem-level pass rates; each problem-level rate is passing valid attempts divided by valid attempts",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "pass-rate",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Eval-level pass rate",
          "tolerance": "A problem attempt passes only when all graded fields satisfy their problem-specific exact-match rules or absolute numeric tolerances.",
          "unit": "percent"
        }
      ],
      "model_ids": [
        "gemini-3-1-pro",
        "gemini-3-5-flash",
        "glm-5-2",
        "gpt-5-2",
        "gpt-5-4",
        "gpt-5-5",
        "gpt-5-6-luna",
        "gpt-5-6-sol",
        "gpt-5-6-terra",
        "grok-4-3"
      ],
      "protocol": {
        "contamination": {
          "notes": "The full reported evaluation includes every release stratum.",
          "reporting_status": "reported",
          "value": "constructively simulated suite with disjoint 10 public, 50 Artificial Analysis, and 69 internal-holdout release strata"
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "problem-specific Python binary checker requiring every graded field to satisfy exact-match or absolute-tolerance constraints"
        },
        "reasoning": {
          "notes": "Effort labels are provider-native configuration labels; no claim is made that equal labels imply equal token budgets across providers.",
          "reporting_status": "reported",
          "value": "provider-reported high reasoning setting"
        },
        "repeats": {
          "notes": "10 independent attempts per model–problem pair were scheduled; invalid container, tooling, provider, or response-format attempts were excluded.",
          "reporting_status": "reported",
          "value": 10
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The report enumerates the initial instructions but does not characterize the setup using a zero/few-shot convention.",
          "reporting_status": "not_reported",
          "value": null
        },
        "statistical": {
          "notes": "Per-problem rates use the final valid attempts; no partial credit. The free-text reasoning field is collected but not graded.",
          "reporting_status": "reported",
          "value": "unweighted mean of 129 per-problem pass rates with a 95% hierarchical bootstrap CI from 20,000 resamples of problems and repeated runs"
        },
        "system_prompt_public": {
          "notes": "The task prompts for ten cases are public, but the full harness system message and 119 held-out prompts are not published.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Runs remained subject to provider and platform behavior.",
          "reporting_status": "reported",
          "value": "No additional uniform wall-clock budget imposed by the harness"
        },
        "token_budget": {
          "notes": "No uniform token budget is reported. Supplementary Table 1 reports observed average tokens only for mainline GPT-family configurations.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "The isolated workspace had no internet access.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Python and R analysis were available in the Linux environment.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "Runs used a Linux Docker container.",
            "reporting_status": "reported",
            "value": true
          },
          "databases": {
            "notes": "Agents were limited to staged local files, installed software, and model-internal knowledge.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Representative installed packages; the paper provides the longer exact list.",
            "reporting_status": "reported",
            "value": [
              "Python/R scientific stacks",
              "PLINK 2.0",
              "bedtools/tabix",
              "pysam/cyvcf2",
              "scanpy/anndata",
              "DESeq2/edgeR/limma"
            ]
          },
          "internet": {
            "notes": "Methods explicitly state that the execution environment had no internet access.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "Agents iteratively inspect files, execute analyses, and return a final JSON object.",
          "reporting_status": "reported",
          "value": "multi-turn agent-container trajectory"
        }
      },
      "results": [
        {
          "ci_high": 2.9,
          "ci_low": 0.5,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-standard-high-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "grok-4-3",
          "n": 129,
          "notes": "Supplementary Table 1 configuration Grok4.3 (high); nominal attempts per problem 10, valid-attempt mean 10.0 and range 10–10; average tokens not reported; problem regimes 0%=92.2%, 0–10%=4.7%, 10–50%=2.3%, ≥50%=0.8%.",
          "status": "verified",
          "value": 1.5
        },
        {
          "ci_high": 5.0,
          "ci_low": 1.6,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-standard-high-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gemini-3-1-pro",
          "n": 129,
          "notes": "Supplementary Table 1 configuration Gemini3.1Pro (high); nominal attempts per problem 10, valid-attempt mean 10.0 and range 10–10; average tokens not reported; problem regimes 0%=81.4%, 0–10%=13.2%, 10–50%=4.7%, ≥50%=0.8%.",
          "status": "verified",
          "value": 3.1
        },
        {
          "ci_high": 7.0,
          "ci_low": 2.6,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-standard-high-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "glm-5-2",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GLM5.2 (high); nominal attempts per problem 10, valid-attempt mean 9.8 and range 8–10; average tokens not reported; problem regimes 0%=77.5%, 0–10%=10.1%, 10–50%=10.9%, ≥50%=1.6%.",
          "status": "verified",
          "value": 4.6
        },
        {
          "ci_high": 11.6,
          "ci_low": 5.1,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-standard-high-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gemini-3-5-flash",
          "n": 129,
          "notes": "Supplementary Table 1 configuration Gemini3.5Flash (high); nominal attempts per problem 10, valid-attempt mean 10.0 and range 10–10; average tokens not reported; problem regimes 0%=70.5%, 0–10%=14.7%, 10–50%=9.3%, ≥50%=5.4%.",
          "status": "verified",
          "value": 8.1
        },
        {
          "ci_high": 5.7,
          "ci_low": 1.6,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-standard-high-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-2",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.2 (high); nominal attempts per problem 10, valid-attempt mean 9.9 and range 9–10; average tokens 23.2k; problem regimes 0%=83.7%, 0–10%=7.8%, 10–50%=7.0%, ≥50%=1.6%.",
          "status": "verified",
          "value": 3.5
        },
        {
          "ci_high": 9.9,
          "ci_low": 4.4,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-standard-high-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-4",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.4 (high); nominal attempts per problem 10, valid-attempt mean 10.0 and range 9–10; average tokens 27.3k; problem regimes 0%=70.5%, 0–10%=11.6%, 10–50%=13.2%, ≥50%=4.7%.",
          "status": "verified",
          "value": 7.0
        },
        {
          "ci_high": 13.1,
          "ci_low": 5.8,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-standard-high-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-5",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.5 (high); nominal attempts per problem 10, valid-attempt mean 10.0 and range 9–10; average tokens 19.8k; problem regimes 0%=70.5%, 0–10%=11.6%, 10–50%=10.9%, ≥50%=7.0%.",
          "status": "verified",
          "value": 9.3
        },
        {
          "ci_high": 11.7,
          "ci_low": 4.8,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-standard-high-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-6-luna",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.6Luna (high); nominal attempts per problem 10, valid-attempt mean 9.8 and range 7–10; average tokens 32.3k; problem regimes 0%=76.7%, 0–10%=8.5%, 10–50%=7.8%, ≥50%=7.0%.",
          "status": "verified",
          "value": 8.0
        },
        {
          "ci_high": 21.2,
          "ci_low": 11.6,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-standard-high-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-6-terra",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.6Terra (high); nominal attempts per problem 10, valid-attempt mean 10.0 and range 9–10; average tokens 22.2k; problem regimes 0%=59.7%, 0–10%=12.4%, 10–50%=11.6%, ≥50%=16.3%.",
          "status": "verified",
          "value": 16.2
        },
        {
          "ci_high": 30.5,
          "ci_low": 18.5,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-standard-high-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-6-sol",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.6Sol (high); nominal attempts per problem 10, valid-attempt mean 10.0 and range 9–10; average tokens 19.5k; problem regimes 0%=51.9%, 0–10%=7.8%, 10–50%=17.8%, ≥50%=22.5%.",
          "status": "verified",
          "value": 24.4
        }
      ],
      "scope": {
        "filter": "Complete formal 129-problem suite, including the 10 public, 50 Artificial Analysis, and 69 internal-holdout problems.",
        "n": 129,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "10 configuration result(s), confidence intervals, repeats, valid-attempt summaries, and environment settings were checked against Methods and Supplementary Table 1.",
        "status": "verified"
      },
      "work_id": "genebench-pro-report",
      "work_version_id": "genebench-pro-report-2026-06-30"
    },
    {
      "benchmark_id": "genebench-pro",
      "benchmark_version": "paper-v1",
      "comparability_group": "genebench-pro-paper-v1-full-standard-low",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "genebench-pro-standard-low-protocol",
          "locator": {
            "note": "Full scope, attempts, invalid-run exclusion, Docker environment, installed tools, no internet, response schema, binary grader, and no uniform harness wall-clock budget.",
            "type": "section",
            "value": "Methods — Evaluation and grading, printed page 15"
          },
          "source_id": "genebench-pro-report",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "genebench-pro-standard-low-results",
          "locator": {
            "note": "Exact configuration labels, pass rates, 95% CI bounds, average tokens where available, valid-attempt summaries, and problem-regime shares.",
            "type": "table",
            "value": "Supplementary Table 1, printed page 21"
          },
          "source_id": "genebench-pro-report",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "genebench-pro-standard-low",
      "metrics": [
        {
          "aggregation": "unweighted mean across 129 problem-level pass rates; each problem-level rate is passing valid attempts divided by valid attempts",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "pass-rate",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Eval-level pass rate",
          "tolerance": "A problem attempt passes only when all graded fields satisfy their problem-specific exact-match rules or absolute numeric tolerances.",
          "unit": "percent"
        }
      ],
      "model_ids": [
        "gpt-5-2",
        "gpt-5-4",
        "gpt-5-5",
        "gpt-5-6-luna",
        "gpt-5-6-sol",
        "gpt-5-6-terra"
      ],
      "protocol": {
        "contamination": {
          "notes": "The full reported evaluation includes every release stratum.",
          "reporting_status": "reported",
          "value": "constructively simulated suite with disjoint 10 public, 50 Artificial Analysis, and 69 internal-holdout release strata"
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "problem-specific Python binary checker requiring every graded field to satisfy exact-match or absolute-tolerance constraints"
        },
        "reasoning": {
          "notes": "Effort labels are provider-native configuration labels; no claim is made that equal labels imply equal token budgets across providers.",
          "reporting_status": "reported",
          "value": "provider-reported low reasoning setting"
        },
        "repeats": {
          "notes": "10 independent attempts per model–problem pair were scheduled; invalid container, tooling, provider, or response-format attempts were excluded.",
          "reporting_status": "reported",
          "value": 10
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The report enumerates the initial instructions but does not characterize the setup using a zero/few-shot convention.",
          "reporting_status": "not_reported",
          "value": null
        },
        "statistical": {
          "notes": "Per-problem rates use the final valid attempts; no partial credit. The free-text reasoning field is collected but not graded.",
          "reporting_status": "reported",
          "value": "unweighted mean of 129 per-problem pass rates with a 95% hierarchical bootstrap CI from 20,000 resamples of problems and repeated runs"
        },
        "system_prompt_public": {
          "notes": "The task prompts for ten cases are public, but the full harness system message and 119 held-out prompts are not published.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Runs remained subject to provider and platform behavior.",
          "reporting_status": "reported",
          "value": "No additional uniform wall-clock budget imposed by the harness"
        },
        "token_budget": {
          "notes": "No uniform token budget is reported. Supplementary Table 1 reports observed average tokens only for mainline GPT-family configurations.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "The isolated workspace had no internet access.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Python and R analysis were available in the Linux environment.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "Runs used a Linux Docker container.",
            "reporting_status": "reported",
            "value": true
          },
          "databases": {
            "notes": "Agents were limited to staged local files, installed software, and model-internal knowledge.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Representative installed packages; the paper provides the longer exact list.",
            "reporting_status": "reported",
            "value": [
              "Python/R scientific stacks",
              "PLINK 2.0",
              "bedtools/tabix",
              "pysam/cyvcf2",
              "scanpy/anndata",
              "DESeq2/edgeR/limma"
            ]
          },
          "internet": {
            "notes": "Methods explicitly state that the execution environment had no internet access.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "Agents iteratively inspect files, execute analyses, and return a final JSON object.",
          "reporting_status": "reported",
          "value": "multi-turn agent-container trajectory"
        }
      },
      "results": [
        {
          "ci_high": 2.0,
          "ci_low": 0.3,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-standard-low-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-2",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.2 (low); nominal attempts per problem 10, valid-attempt mean 10.0 and range 9–10; average tokens 7.0k; problem regimes 0%=91.5%, 0–10%=6.2%, 10–50%=2.3%, ≥50%=0.0%.",
          "status": "verified",
          "value": 1.1
        },
        {
          "ci_high": 5.0,
          "ci_low": 1.4,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-standard-low-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-4",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.4 (low); nominal attempts per problem 10, valid-attempt mean 10.0 and range 9–10; average tokens 8.2k; problem regimes 0%=85.3%, 0–10%=7.0%, 10–50%=7.0%, ≥50%=0.8%.",
          "status": "verified",
          "value": 3.0
        },
        {
          "ci_high": 4.1,
          "ci_low": 1.0,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-standard-low-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-5",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.5 (low); nominal attempts per problem 10, valid-attempt mean 9.9 and range 8–10; average tokens 2.8k; problem regimes 0%=87.6%, 0–10%=6.2%, 10–50%=4.7%, ≥50%=1.6%.",
          "status": "verified",
          "value": 2.4
        },
        {
          "ci_high": 4.1,
          "ci_low": 0.9,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-standard-low-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-6-luna",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.6Luna (low); nominal attempts per problem 10, valid-attempt mean 10.0 and range 10–10; average tokens 3.6k; problem regimes 0%=89.1%, 0–10%=5.4%, 10–50%=4.7%, ≥50%=0.8%.",
          "status": "verified",
          "value": 2.3
        },
        {
          "ci_high": 9.5,
          "ci_low": 3.9,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-standard-low-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-6-terra",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.6Terra (low); nominal attempts per problem 10, valid-attempt mean 10.0 and range 9–10; average tokens 5.5k; problem regimes 0%=75.2%, 0–10%=10.1%, 10–50%=10.9%, ≥50%=3.9%.",
          "status": "verified",
          "value": 6.5
        },
        {
          "ci_high": 19.0,
          "ci_low": 10.1,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-standard-low-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-6-sol",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.6Sol (low); nominal attempts per problem 10, valid-attempt mean 10.0 and range 9–10; average tokens 5.6k; problem regimes 0%=58.9%, 0–10%=12.4%, 10–50%=17.1%, ≥50%=11.6%.",
          "status": "verified",
          "value": 14.4
        }
      ],
      "scope": {
        "filter": "Complete formal 129-problem suite, including the 10 public, 50 Artificial Analysis, and 69 internal-holdout problems.",
        "n": 129,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "6 configuration result(s), confidence intervals, repeats, valid-attempt summaries, and environment settings were checked against Methods and Supplementary Table 1.",
        "status": "verified"
      },
      "work_id": "genebench-pro-report",
      "work_version_id": "genebench-pro-report-2026-06-30"
    },
    {
      "benchmark_id": "genebench-pro",
      "benchmark_version": "paper-v1",
      "comparability_group": "genebench-pro-paper-v1-full-standard-max",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "genebench-pro-standard-max-protocol",
          "locator": {
            "note": "Full scope, attempts, invalid-run exclusion, Docker environment, installed tools, no internet, response schema, binary grader, and no uniform harness wall-clock budget.",
            "type": "section",
            "value": "Methods — Evaluation and grading, printed page 15"
          },
          "source_id": "genebench-pro-report",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "genebench-pro-standard-max-results",
          "locator": {
            "note": "Exact configuration labels, pass rates, 95% CI bounds, average tokens where available, valid-attempt summaries, and problem-regime shares.",
            "type": "table",
            "value": "Supplementary Table 1, printed page 21"
          },
          "source_id": "genebench-pro-report",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "genebench-pro-standard-max",
      "metrics": [
        {
          "aggregation": "unweighted mean across 129 problem-level pass rates; each problem-level rate is passing valid attempts divided by valid attempts",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "pass-rate",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Eval-level pass rate",
          "tolerance": "A problem attempt passes only when all graded fields satisfy their problem-specific exact-match rules or absolute numeric tolerances.",
          "unit": "percent"
        }
      ],
      "model_ids": [
        "gpt-5-6-luna",
        "gpt-5-6-sol",
        "gpt-5-6-terra"
      ],
      "protocol": {
        "contamination": {
          "notes": "The full reported evaluation includes every release stratum.",
          "reporting_status": "reported",
          "value": "constructively simulated suite with disjoint 10 public, 50 Artificial Analysis, and 69 internal-holdout release strata"
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "problem-specific Python binary checker requiring every graded field to satisfy exact-match or absolute-tolerance constraints"
        },
        "reasoning": {
          "notes": "Effort labels are provider-native configuration labels; no claim is made that equal labels imply equal token budgets across providers.",
          "reporting_status": "reported",
          "value": "provider-reported max reasoning setting"
        },
        "repeats": {
          "notes": "10 independent attempts per model–problem pair were scheduled; invalid container, tooling, provider, or response-format attempts were excluded.",
          "reporting_status": "reported",
          "value": 10
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The report enumerates the initial instructions but does not characterize the setup using a zero/few-shot convention.",
          "reporting_status": "not_reported",
          "value": null
        },
        "statistical": {
          "notes": "Per-problem rates use the final valid attempts; no partial credit. The free-text reasoning field is collected but not graded.",
          "reporting_status": "reported",
          "value": "unweighted mean of 129 per-problem pass rates with a 95% hierarchical bootstrap CI from 20,000 resamples of problems and repeated runs"
        },
        "system_prompt_public": {
          "notes": "The task prompts for ten cases are public, but the full harness system message and 119 held-out prompts are not published.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Runs remained subject to provider and platform behavior.",
          "reporting_status": "reported",
          "value": "No additional uniform wall-clock budget imposed by the harness"
        },
        "token_budget": {
          "notes": "No uniform token budget is reported. Supplementary Table 1 reports observed average tokens only for mainline GPT-family configurations.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "The isolated workspace had no internet access.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Python and R analysis were available in the Linux environment.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "Runs used a Linux Docker container.",
            "reporting_status": "reported",
            "value": true
          },
          "databases": {
            "notes": "Agents were limited to staged local files, installed software, and model-internal knowledge.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Representative installed packages; the paper provides the longer exact list.",
            "reporting_status": "reported",
            "value": [
              "Python/R scientific stacks",
              "PLINK 2.0",
              "bedtools/tabix",
              "pysam/cyvcf2",
              "scanpy/anndata",
              "DESeq2/edgeR/limma"
            ]
          },
          "internet": {
            "notes": "Methods explicitly state that the execution environment had no internet access.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "Agents iteratively inspect files, execute analyses, and return a final JSON object.",
          "reporting_status": "reported",
          "value": "multi-turn agent-container trajectory"
        }
      },
      "results": [
        {
          "ci_high": 21.7,
          "ci_low": 11.6,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-standard-max-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-6-luna",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.6Luna (max); nominal attempts per problem 10, valid-attempt mean 9.6 and range 2–10; average tokens 118.2k; problem regimes 0%=64.3%, 0–10%=4.7%, 10–50%=16.3%, ≥50%=14.7%.",
          "status": "verified",
          "value": 16.5
        },
        {
          "ci_high": 29.2,
          "ci_low": 17.8,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-standard-max-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-6-terra",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.6Terra (max); nominal attempts per problem 10, valid-attempt mean 10.0 and range 10–10; average tokens 54.3k; problem regimes 0%=49.6%, 0–10%=9.3%, 10–50%=19.4%, ≥50%=21.7%.",
          "status": "verified",
          "value": 23.3
        },
        {
          "ci_high": 35.1,
          "ci_low": 22.5,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-standard-max-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-6-sol",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.6Sol (max); nominal attempts per problem 10, valid-attempt mean 10.0 and range 9–10; average tokens 33.2k; problem regimes 0%=45.7%, 0–10%=10.1%, 10–50%=14.0%, ≥50%=30.2%.",
          "status": "verified",
          "value": 28.7
        }
      ],
      "scope": {
        "filter": "Complete formal 129-problem suite, including the 10 public, 50 Artificial Analysis, and 69 internal-holdout problems.",
        "n": 129,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "3 configuration result(s), confidence intervals, repeats, valid-attempt summaries, and environment settings were checked against Methods and Supplementary Table 1.",
        "status": "verified"
      },
      "work_id": "genebench-pro-report",
      "work_version_id": "genebench-pro-report-2026-06-30"
    },
    {
      "benchmark_id": "genebench-pro",
      "benchmark_version": "paper-v1",
      "comparability_group": "genebench-pro-paper-v1-full-standard-medium",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "genebench-pro-standard-medium-protocol",
          "locator": {
            "note": "Full scope, attempts, invalid-run exclusion, Docker environment, installed tools, no internet, response schema, binary grader, and no uniform harness wall-clock budget.",
            "type": "section",
            "value": "Methods — Evaluation and grading, printed page 15"
          },
          "source_id": "genebench-pro-report",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "genebench-pro-standard-medium-results",
          "locator": {
            "note": "Exact configuration labels, pass rates, 95% CI bounds, average tokens where available, valid-attempt summaries, and problem-regime shares.",
            "type": "table",
            "value": "Supplementary Table 1, printed page 21"
          },
          "source_id": "genebench-pro-report",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "genebench-pro-standard-medium",
      "metrics": [
        {
          "aggregation": "unweighted mean across 129 problem-level pass rates; each problem-level rate is passing valid attempts divided by valid attempts",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "pass-rate",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Eval-level pass rate",
          "tolerance": "A problem attempt passes only when all graded fields satisfy their problem-specific exact-match rules or absolute numeric tolerances.",
          "unit": "percent"
        }
      ],
      "model_ids": [
        "gpt-5-2",
        "gpt-5-4",
        "gpt-5-5",
        "gpt-5-6-luna",
        "gpt-5-6-sol",
        "gpt-5-6-terra"
      ],
      "protocol": {
        "contamination": {
          "notes": "The full reported evaluation includes every release stratum.",
          "reporting_status": "reported",
          "value": "constructively simulated suite with disjoint 10 public, 50 Artificial Analysis, and 69 internal-holdout release strata"
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "problem-specific Python binary checker requiring every graded field to satisfy exact-match or absolute-tolerance constraints"
        },
        "reasoning": {
          "notes": "Effort labels are provider-native configuration labels; no claim is made that equal labels imply equal token budgets across providers.",
          "reporting_status": "reported",
          "value": "provider-reported medium reasoning setting"
        },
        "repeats": {
          "notes": "10 independent attempts per model–problem pair were scheduled; invalid container, tooling, provider, or response-format attempts were excluded.",
          "reporting_status": "reported",
          "value": 10
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The report enumerates the initial instructions but does not characterize the setup using a zero/few-shot convention.",
          "reporting_status": "not_reported",
          "value": null
        },
        "statistical": {
          "notes": "Per-problem rates use the final valid attempts; no partial credit. The free-text reasoning field is collected but not graded.",
          "reporting_status": "reported",
          "value": "unweighted mean of 129 per-problem pass rates with a 95% hierarchical bootstrap CI from 20,000 resamples of problems and repeated runs"
        },
        "system_prompt_public": {
          "notes": "The task prompts for ten cases are public, but the full harness system message and 119 held-out prompts are not published.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Runs remained subject to provider and platform behavior.",
          "reporting_status": "reported",
          "value": "No additional uniform wall-clock budget imposed by the harness"
        },
        "token_budget": {
          "notes": "No uniform token budget is reported. Supplementary Table 1 reports observed average tokens only for mainline GPT-family configurations.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "The isolated workspace had no internet access.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Python and R analysis were available in the Linux environment.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "Runs used a Linux Docker container.",
            "reporting_status": "reported",
            "value": true
          },
          "databases": {
            "notes": "Agents were limited to staged local files, installed software, and model-internal knowledge.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Representative installed packages; the paper provides the longer exact list.",
            "reporting_status": "reported",
            "value": [
              "Python/R scientific stacks",
              "PLINK 2.0",
              "bedtools/tabix",
              "pysam/cyvcf2",
              "scanpy/anndata",
              "DESeq2/edgeR/limma"
            ]
          },
          "internet": {
            "notes": "Methods explicitly state that the execution environment had no internet access.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "Agents iteratively inspect files, execute analyses, and return a final JSON object.",
          "reporting_status": "reported",
          "value": "multi-turn agent-container trajectory"
        }
      },
      "results": [
        {
          "ci_high": 4.2,
          "ci_low": 1.0,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-standard-medium-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-2",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.2 (medium); nominal attempts per problem 10, valid-attempt mean 10.0 and range 9–10; average tokens 18.1k; problem regimes 0%=86.0%, 0–10%=10.1%, 10–50%=3.1%, ≥50%=0.8%.",
          "status": "verified",
          "value": 2.4
        },
        {
          "ci_high": 7.7,
          "ci_low": 2.7,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-standard-medium-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-4",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.4 (medium); nominal attempts per problem 10, valid-attempt mean 10.0 and range 9–10; average tokens 18.6k; problem regimes 0%=74.4%, 0–10%=17.1%, 10–50%=7.0%, ≥50%=1.6%.",
          "status": "verified",
          "value": 5.0
        },
        {
          "ci_high": 8.8,
          "ci_low": 3.3,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-standard-medium-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-5",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.5 (medium); nominal attempts per problem 10, valid-attempt mean 10.0 and range 9–10; average tokens 10.6k; problem regimes 0%=79.1%, 0–10%=7.8%, 10–50%=9.3%, ≥50%=3.9%.",
          "status": "verified",
          "value": 5.9
        },
        {
          "ci_high": 7.6,
          "ci_low": 2.3,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-standard-medium-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-6-luna",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.6Luna (medium); nominal attempts per problem 10, valid-attempt mean 9.9 and range 8–10; average tokens 15.6k; problem regimes 0%=83.7%, 0–10%=7.0%, 10–50%=4.7%, ≥50%=4.7%.",
          "status": "verified",
          "value": 4.7
        },
        {
          "ci_high": 18.2,
          "ci_low": 9.3,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-standard-medium-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-6-terra",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.6Terra (medium); nominal attempts per problem 10, valid-attempt mean 10.0 and range 10–10; average tokens 15.9k; problem regimes 0%=65.9%, 0–10%=8.5%, 10–50%=12.4%, ≥50%=13.2%.",
          "status": "verified",
          "value": 13.6
        },
        {
          "ci_high": 28.4,
          "ci_low": 16.9,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-standard-medium-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-6-sol",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.6Sol (medium); nominal attempts per problem 10, valid-attempt mean 10.0 and range 9–10; average tokens 14.4k; problem regimes 0%=53.5%, 0–10%=7.8%, 10–50%=16.3%, ≥50%=22.5%.",
          "status": "verified",
          "value": 22.5
        }
      ],
      "scope": {
        "filter": "Complete formal 129-problem suite, including the 10 public, 50 Artificial Analysis, and 69 internal-holdout problems.",
        "n": 129,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "6 configuration result(s), confidence intervals, repeats, valid-attempt summaries, and environment settings were checked against Methods and Supplementary Table 1.",
        "status": "verified"
      },
      "work_id": "genebench-pro-report",
      "work_version_id": "genebench-pro-report-2026-06-30"
    },
    {
      "benchmark_id": "genebench-pro",
      "benchmark_version": "paper-v1",
      "comparability_group": "genebench-pro-paper-v1-full-standard-none",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "genebench-pro-standard-none-protocol",
          "locator": {
            "note": "Full scope, attempts, invalid-run exclusion, Docker environment, installed tools, no internet, response schema, binary grader, and no uniform harness wall-clock budget.",
            "type": "section",
            "value": "Methods — Evaluation and grading, printed page 15"
          },
          "source_id": "genebench-pro-report",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "genebench-pro-standard-none-results",
          "locator": {
            "note": "Exact configuration labels, pass rates, 95% CI bounds, average tokens where available, valid-attempt summaries, and problem-regime shares.",
            "type": "table",
            "value": "Supplementary Table 1, printed page 21"
          },
          "source_id": "genebench-pro-report",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "genebench-pro-standard-none",
      "metrics": [
        {
          "aggregation": "unweighted mean across 129 problem-level pass rates; each problem-level rate is passing valid attempts divided by valid attempts",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "pass-rate",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Eval-level pass rate",
          "tolerance": "A problem attempt passes only when all graded fields satisfy their problem-specific exact-match rules or absolute numeric tolerances.",
          "unit": "percent"
        }
      ],
      "model_ids": [
        "gpt-5-2",
        "gpt-5-4",
        "gpt-5-5",
        "gpt-5-6-luna",
        "gpt-5-6-sol",
        "gpt-5-6-terra"
      ],
      "protocol": {
        "contamination": {
          "notes": "The full reported evaluation includes every release stratum.",
          "reporting_status": "reported",
          "value": "constructively simulated suite with disjoint 10 public, 50 Artificial Analysis, and 69 internal-holdout release strata"
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "problem-specific Python binary checker requiring every graded field to satisfy exact-match or absolute-tolerance constraints"
        },
        "reasoning": {
          "notes": "Effort labels are provider-native configuration labels; no claim is made that equal labels imply equal token budgets across providers.",
          "reporting_status": "reported",
          "value": "provider-reported none reasoning setting"
        },
        "repeats": {
          "notes": "10 independent attempts per model–problem pair were scheduled; invalid container, tooling, provider, or response-format attempts were excluded.",
          "reporting_status": "reported",
          "value": 10
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The report enumerates the initial instructions but does not characterize the setup using a zero/few-shot convention.",
          "reporting_status": "not_reported",
          "value": null
        },
        "statistical": {
          "notes": "Per-problem rates use the final valid attempts; no partial credit. The free-text reasoning field is collected but not graded.",
          "reporting_status": "reported",
          "value": "unweighted mean of 129 per-problem pass rates with a 95% hierarchical bootstrap CI from 20,000 resamples of problems and repeated runs"
        },
        "system_prompt_public": {
          "notes": "The task prompts for ten cases are public, but the full harness system message and 119 held-out prompts are not published.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Runs remained subject to provider and platform behavior.",
          "reporting_status": "reported",
          "value": "No additional uniform wall-clock budget imposed by the harness"
        },
        "token_budget": {
          "notes": "No uniform token budget is reported. Supplementary Table 1 reports observed average tokens only for mainline GPT-family configurations.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "The isolated workspace had no internet access.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Python and R analysis were available in the Linux environment.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "Runs used a Linux Docker container.",
            "reporting_status": "reported",
            "value": true
          },
          "databases": {
            "notes": "Agents were limited to staged local files, installed software, and model-internal knowledge.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Representative installed packages; the paper provides the longer exact list.",
            "reporting_status": "reported",
            "value": [
              "Python/R scientific stacks",
              "PLINK 2.0",
              "bedtools/tabix",
              "pysam/cyvcf2",
              "scanpy/anndata",
              "DESeq2/edgeR/limma"
            ]
          },
          "internet": {
            "notes": "Methods explicitly state that the execution environment had no internet access.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "Agents iteratively inspect files, execute analyses, and return a final JSON object.",
          "reporting_status": "reported",
          "value": "multi-turn agent-container trajectory"
        }
      },
      "results": [
        {
          "ci_high": 1.5,
          "ci_low": 0.0,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-standard-none-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-2",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.2 (none); nominal attempts per problem 10, valid-attempt mean 10.0 and range 9–10; average tokens 1.1k; problem regimes 0%=96.9%, 0–10%=2.3%, 10–50%=0.8%, ≥50%=0.0%.",
          "status": "verified",
          "value": 0.5
        },
        {
          "ci_high": 1.9,
          "ci_low": 0.2,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-standard-none-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-4",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.4 (none); nominal attempts per problem 10, valid-attempt mean 10.0 and range 9–10; average tokens 2.1k; problem regimes 0%=94.6%, 0–10%=3.9%, 10–50%=1.6%, ≥50%=0.0%.",
          "status": "verified",
          "value": 0.9
        },
        {
          "ci_high": 1.8,
          "ci_low": 0.1,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-standard-none-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-5",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.5 (none); nominal attempts per problem 10, valid-attempt mean 10.0 and range 9–10; average tokens 1.1k; problem regimes 0%=96.1%, 0–10%=2.3%, 10–50%=1.6%, ≥50%=0.0%.",
          "status": "verified",
          "value": 0.8
        },
        {
          "ci_high": 1.6,
          "ci_low": 0.2,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-standard-none-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-6-luna",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.6Luna (none); nominal attempts per problem 10, valid-attempt mean 10.0 and range 9–10; average tokens 975; problem regimes 0%=93.8%, 0–10%=4.7%, 10–50%=1.6%, ≥50%=0.0%.",
          "status": "verified",
          "value": 0.8
        },
        {
          "ci_high": 2.3,
          "ci_low": 0.1,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-standard-none-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-6-terra",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.6Terra (none); nominal attempts per problem 10, valid-attempt mean 10.0 and range 10–10; average tokens 930; problem regimes 0%=95.3%, 0–10%=3.1%, 10–50%=0.8%, ≥50%=0.8%.",
          "status": "verified",
          "value": 1.0
        },
        {
          "ci_high": 5.9,
          "ci_low": 1.9,
          "confidence": "high",
          "evidence_ids": [
            "genebench-pro-standard-none-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-6-sol",
          "n": 129,
          "notes": "Supplementary Table 1 configuration GPT-5.6Sol (none); nominal attempts per problem 10, valid-attempt mean 10.0 and range 9–10; average tokens 1.4k; problem regimes 0%=82.2%, 0–10%=9.3%, 10–50%=7.0%, ≥50%=1.6%.",
          "status": "verified",
          "value": 3.7
        }
      ],
      "scope": {
        "filter": "Complete formal 129-problem suite, including the 10 public, 50 Artificial Analysis, and 69 internal-holdout problems.",
        "n": 129,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "6 configuration result(s), confidence intervals, repeats, valid-attempt summaries, and environment settings were checked against Methods and Supplementary Table 1.",
        "status": "verified"
      },
      "work_id": "genebench-pro-report",
      "work_version_id": "genebench-pro-report-2026-06-30"
    },
    {
      "benchmark_id": "genomic-benchmarks",
      "benchmark_version": "package-1.0.0-snapshot",
      "comparability_group": "genomic-benchmarks-creator-native",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "genomic-benchmarks-creator-full-protocol-evidence",
          "locator": {
            "note": "Lists all nine datasets, train/test design, CNN workflows, and accuracy/F1 results.",
            "type": "table",
            "value": "Methods and Tables 1-2"
          },
          "source_id": "genomic-benchmarks-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/protocol",
            "/metrics"
          ]
        }
      ],
      "id": "genomic-benchmarks-creator-full",
      "metrics": [
        {
          "aggregation": "held-out test sequences per dataset",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "classification-accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Accuracy",
          "tolerance": null,
          "unit": "percent"
        },
        {
          "aggregation": "held-out test sequences per dataset",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "f1-score",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "F1 score",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [],
      "protocol": {
        "contamination": {
          "notes": "Version 0 dataset metadata is retained.",
          "reporting_status": "reported",
          "value": "Dataset-specific train/test splits; duplicate and background-generation controls follow each construction notebook."
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "deterministic classification scorer"
        },
        "reasoning": {
          "notes": "Supervised DNA classification; no prompting.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "repeats": {
          "notes": "A common repeat count is not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "A common seed is not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Supervised DNA classification; no prompting.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "statistical": {
          "notes": "No collection-wide aggregate is defined.",
          "reporting_status": "reported",
          "value": "Accuracy and F1 are reported independently for each dataset and framework."
        },
        "system_prompt_public": {
          "notes": "Supervised DNA classification; no prompting.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "temperature": {
          "notes": "Supervised DNA classification; no prompting.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "time_budget": {
          "notes": "No common runtime budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Supervised DNA classification; no prompting.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Supervised DNA classification; no prompting.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "code_execution": {
            "notes": "Training code is the benchmark implementation.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "container": {
            "notes": "No common container is prescribed.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Supervised DNA classification; no prompting.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "external_tools": {
            "notes": "Each dataset uses its published train/test split.",
            "reporting_status": "reported",
            "value": "TensorFlow and PyTorch CNN baseline workflows"
          },
          "internet": {
            "notes": "Supervised DNA classification; no prompting.",
            "reporting_status": "not_applicable",
            "value": null
          }
        },
        "turns": {
          "notes": "Supervised DNA classification; no prompting.",
          "reporting_status": "not_applicable",
          "value": null
        }
      },
      "results": [],
      "scope": {
        "filter": "All nine datasets listed in creator-paper Table 1.",
        "n": 9,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Nine-dataset scope and both reported metrics are verified; per-dataset table values remain downloadable from the source.",
        "status": "verified"
      },
      "work_id": "genomic-benchmarks-paper",
      "work_version_id": "genomic-benchmarks-paper-2023-05-01"
    },
    {
      "benchmark_id": "guacamol",
      "benchmark_version": "suite-v2",
      "comparability_group": "guacamol-v2-task-native",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "guacamol-creator-full-protocol-evidence",
          "locator": {
            "note": "Defines both modes, data standardization, baseline generators, and scoring. The pinned implementation establishes the current twenty-problem v2 list.",
            "type": "section",
            "value": "Sections 2-4 and Tables 1-2; official benchmark_suites.py v2"
          },
          "source_id": "guacamol-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/protocol",
            "/metrics"
          ]
        }
      ],
      "id": "guacamol-creator-full",
      "metrics": [
        {
          "aggregation": "generated sample",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "validity-score",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Validity",
          "tolerance": null,
          "unit": "proportion"
        },
        {
          "aggregation": "generated sample",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "uniqueness-score",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Uniqueness",
          "tolerance": null,
          "unit": "proportion"
        },
        {
          "aggregation": "generated sample",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "novelty-score",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Novelty",
          "tolerance": null,
          "unit": "proportion"
        },
        {
          "aggregation": "physicochemical descriptor distributions",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "kl-divergence-score",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "KL divergence score",
          "tolerance": null,
          "unit": "normalized score"
        },
        {
          "aggregation": "generated versus reference distribution",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "frechet-chemnet-distance-score",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Fréchet ChemNet Distance score",
          "tolerance": null,
          "unit": "normalized score"
        },
        {
          "aggregation": "benchmark-specific top generated molecules",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "goal-directed-score",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Goal-directed benchmark score",
          "tolerance": null,
          "unit": "normalized score"
        }
      ],
      "model_ids": [],
      "protocol": {
        "contamination": {
          "notes": "Published file hashes support reproducibility.",
          "reporting_status": "reported",
          "value": "Standardized ChEMBL training data exclude a designated holdout set and highly similar molecules."
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "deterministic chemistry scoring functions"
        },
        "reasoning": {
          "notes": "Generative chemistry benchmark; no prompting protocol is prescribed.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "repeats": {
          "notes": "Repeat count is model-specific.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "No universal seed is prescribed across generators.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Generative chemistry benchmark; no prompting protocol is prescribed.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "statistical": {
          "notes": "Distribution learning samples 10000 molecules by default.",
          "reporting_status": "reported",
          "value": "Each formal benchmark returns a normalized score; no registry-wide sum is created."
        },
        "system_prompt_public": {
          "notes": "Generative chemistry benchmark; no prompting protocol is prescribed.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "temperature": {
          "notes": "Sampling controls are model-specific.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "No universal generation-time budget is specified.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Generative chemistry benchmark; no prompting protocol is prescribed.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Generative chemistry benchmark; no prompting protocol is prescribed.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "code_execution": {
            "notes": "Generator execution is the evaluated system.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "container": {
            "notes": "The companion baseline repository provides a reproducible environment.",
            "reporting_status": "reported",
            "value": "official Dockerfile"
          },
          "databases": {
            "notes": "Generative chemistry benchmark; no prompting protocol is prescribed.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "external_tools": {
            "notes": "Dependency versions affect standardization and scoring.",
            "reporting_status": "reported",
            "value": "RDKit 2018.09.1 or newer and FCD 1.1"
          },
          "internet": {
            "notes": "Generative chemistry benchmark; no prompting protocol is prescribed.",
            "reporting_status": "not_applicable",
            "value": null
          }
        },
        "turns": {
          "notes": "Generative chemistry benchmark; no prompting protocol is prescribed.",
          "reporting_status": "not_applicable",
          "value": null
        }
      },
      "results": [],
      "scope": {
        "filter": "Five distribution-learning benchmarks and twenty goal-directed v2 problems.",
        "n": 25,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Full v2 scope, data holdout, default sample count, dependencies, and score direction are verified.",
        "status": "verified"
      },
      "work_id": "guacamol-paper",
      "work_version_id": "guacamol-paper-2019-03-19"
    },
    {
      "benchmark_id": "lab-bench-cloning-scenarios",
      "benchmark_version": null,
      "comparability_group": "lab-bench-cloning-scenarios-anthropic-10shot-no-tools",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-cloning-scenarios-anthropic-sonnet45-system-card-protocol",
          "locator": {
            "note": "Four selected tracks, 10-shot prompting, no search/bioinformatics tools, threshold, and unreported scope/repeats/grader.",
            "type": "section",
            "value": "Section 9.2.4.4, printed pages 132–133"
          },
          "source_id": "anthropic-sonnet-4-5-system-card",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-cloning-scenarios-anthropic-sonnet45-system-card-results",
          "locator": {
            "note": "Exact point labels. Protocol Sonnet 4/4.5 labels are represented in the Claude for Life Sciences run to avoid duplicate result rows.",
            "type": "figure",
            "value": "Figure 9.2.4.4.A, printed page 133"
          },
          "source_id": "anthropic-sonnet-4-5-system-card",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-cloning-scenarios-anthropic-sonnet45-system-card",
      "metrics": [
        {
          "aggregation": null,
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "LAB-Bench score",
          "tolerance": null,
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "claude-opus-4",
        "claude-opus-4-1",
        "claude-sonnet-4",
        "claude-sonnet-4-5"
      ],
      "protocol": {
        "contamination": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": null,
          "model": null,
          "reporting_status": "not_reported",
          "type": null
        },
        "reasoning": {
          "notes": "Thinking/effort configuration is not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "repeats": {
          "notes": "The number of repeats underlying the plotted error bars is not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Figure title and caption specify 10-shot prompting.",
          "reporting_status": "reported",
          "value": 10
        },
        "statistical": {
          "notes": "The figure does not give numeric interval bounds.",
          "reporting_status": "reported",
          "value": "reported point estimate with plotted error bars"
        },
        "system_prompt_public": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "The system card says search and bioinformatics tools were not included in the testing environment.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "The system card says search and bioinformatics tools were not included in the testing environment.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "The system card says search and bioinformatics tools were not included in the testing environment.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "The system card says search and bioinformatics tools were not included in the testing environment.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "The system card says search and bioinformatics tools were not included in the testing environment.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "The system card says search and bioinformatics tools were not included in the testing environment.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "Multiple-choice evaluation.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-cloning-scenarios-anthropic-sonnet45-system-card-results"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-sonnet-4",
          "n": null,
          "notes": "Point label transcribed from Figure 9.2.4.4.A; numeric error-bar bounds are not reported.",
          "status": "verified",
          "value": 0.485
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-cloning-scenarios-anthropic-sonnet45-system-card-results"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-opus-4",
          "n": null,
          "notes": "Point label transcribed from Figure 9.2.4.4.A; numeric error-bar bounds are not reported.",
          "status": "verified",
          "value": 0.545
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-cloning-scenarios-anthropic-sonnet45-system-card-results"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-opus-4-1",
          "n": null,
          "notes": "Point label transcribed from Figure 9.2.4.4.A; numeric error-bar bounds are not reported.",
          "status": "verified",
          "value": 0.758
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-cloning-scenarios-anthropic-sonnet45-system-card-results"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-sonnet-4-5",
          "n": null,
          "notes": "Point label transcribed from Figure 9.2.4.4.A; numeric error-bar bounds are not reported.",
          "status": "verified",
          "value": 0.667
        }
      ],
      "scope": {
        "filter": "lab-bench-cloning-scenarios track; benchmark snapshot, public/private scope, and realized n are not reported.",
        "n": null,
        "reporting_status": "not_reported",
        "selection": null,
        "subset_id": null,
        "type": "track"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Protocol, exact point labels, and missing n/repeat details were checked on printed pages 132–133.",
        "status": "verified"
      },
      "work_id": "anthropic-sonnet-4-5-system-card",
      "work_version_id": "anthropic-sonnet-4-5-system-card-2025-09-29"
    },
    {
      "benchmark_id": "lab-bench-cloning-scenarios",
      "benchmark_version": "paper-v3",
      "comparability_group": "lab-bench-cloning-scenarios-paper-v3-zero-shot-cot-no-tools",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-cloning-scenarios-creator-mcq-protocol",
          "locator": {
            "note": "Full scope, zero-shot chain-of-thought prompt, no tools, three runs, exact metrics, parser, and model identities.",
            "type": "section",
            "value": "Appendix D.1–D.4; Table 1 and Appendix Table 6"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-cloning-scenarios-creator-mcq-results",
          "locator": {
            "note": "Accuracy, precision, and coverage. Human row excluded.",
            "type": "table",
            "value": "Tables 2–4, row CloningScenarios"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-cloning-scenarios-creator-mcq",
      "metrics": [
        {
          "aggregation": "correct / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Accuracy",
          "tolerance": "exact choice",
          "unit": "proportion"
        },
        {
          "aggregation": "correct / attempted questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Precision (selective accuracy)",
          "tolerance": "insufficient-information responses excluded from denominator",
          "unit": "proportion"
        },
        {
          "aggregation": "attempted / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "coverage",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Coverage",
          "tolerance": "insufficient-information responses treated as not attempted",
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "lab-bench-claude-3-5-sonnet-20240620",
        "lab-bench-claude-3-haiku-20240307",
        "lab-bench-claude-3-opus-20240229",
        "lab-bench-gemini-1-5-pro-001",
        "lab-bench-gpt-4-turbo-unversioned",
        "lab-bench-gpt-4o-unversioned"
      ],
      "protocol": {
        "contamination": {
          "notes": "Main results combine both splits.",
          "reporting_status": "reported",
          "value": "full pre-release snapshot with 80% later public and 20% private contamination-monitoring splits"
        },
        "grader": {
          "human_review": false,
          "model": "Claude 2 fallback parser",
          "reporting_status": "reported",
          "type": "regex multiple-choice parser with Claude 2 fallback"
        },
        "reasoning": {
          "notes": "Prompt includes Think step by step.",
          "reporting_status": "reported",
          "value": "zero-shot chain-of-thought"
        },
        "repeats": {
          "notes": "Tables 2–4 average three runs per model.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public zero-shot prompt.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "No numeric confidence intervals are reported in Tables 2–4.",
          "reporting_status": "reported",
          "value": "mean across three runs per model"
        },
        "system_prompt_public": {
          "notes": "Appendix D.1 publishes the complete prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No common token budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One multiple-choice completion per question.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-cloning-scenarios-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 41,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.28
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-cloning-scenarios-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 41,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.54
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-cloning-scenarios-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 41,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.52
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-cloning-scenarios-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 41,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.27
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-cloning-scenarios-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 41,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.41
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-cloning-scenarios-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 41,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.65
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-cloning-scenarios-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 41,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.1
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-cloning-scenarios-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 41,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.33
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-cloning-scenarios-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 41,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.3
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-cloning-scenarios-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 41,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.28
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-cloning-scenarios-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 41,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.37
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-cloning-scenarios-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 41,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.77
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-cloning-scenarios-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 41,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.09
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-cloning-scenarios-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 41,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.36
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-cloning-scenarios-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 41,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.31
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-cloning-scenarios-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 41,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.26
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-cloning-scenarios-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 41,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.29
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-cloning-scenarios-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 41,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.9
        }
      ],
      "scope": {
        "filter": "Complete creator snapshot, combining public and private splits.",
        "n": 41,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "All registered values were parsed from creator Tables 2–4 and independently checked against the HTML table rows.",
        "status": "verified"
      },
      "work_id": "lab-bench-paper",
      "work_version_id": "lab-bench-paper-2024-07-14"
    },
    {
      "benchmark_id": "lab-bench-cloning-scenarios",
      "benchmark_version": "paper-v3",
      "comparability_group": "lab-bench-cloning-scenarios-paper-v3-llama-context-limited",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-cloning-scenarios-creator-mcq-llama-context-protocol",
          "locator": {
            "note": "Llama context handling, 25 prompted items, 16 insufficient-information treatments, three-run metrics.",
            "type": "section",
            "value": "Appendix D.2–D.4"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-cloning-scenarios-creator-mcq-llama-context-results",
          "locator": {
            "note": "Accuracy, precision, and coverage.",
            "type": "table",
            "value": "Tables 2–4, CloningScenarios row, Meta-Llama-3-70B-Instruct column"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-cloning-scenarios-creator-mcq-llama-context",
      "metrics": [
        {
          "aggregation": "correct / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Accuracy",
          "tolerance": "exact choice",
          "unit": "proportion"
        },
        {
          "aggregation": "correct / attempted questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Precision (selective accuracy)",
          "tolerance": "insufficient-information responses excluded from denominator",
          "unit": "proportion"
        },
        {
          "aggregation": "attempted / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "coverage",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Coverage",
          "tolerance": "insufficient-information responses treated as not attempted",
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "lab-bench-meta-llama-3-70b-instruct"
      ],
      "protocol": {
        "contamination": {
          "notes": "Main results combine both splits.",
          "reporting_status": "reported",
          "value": "full pre-release snapshot with 80% later public and 20% private contamination-monitoring splits"
        },
        "grader": {
          "human_review": false,
          "model": "Claude 2 fallback parser",
          "reporting_status": "reported",
          "type": "regex multiple-choice parser with Claude 2 fallback"
        },
        "reasoning": {
          "notes": "Prompt includes Think step by step.",
          "reporting_status": "reported",
          "value": "zero-shot chain-of-thought"
        },
        "repeats": {
          "notes": "Tables 2–4 average three runs per model.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public zero-shot prompt.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "No numeric confidence intervals are reported in Tables 2–4.",
          "reporting_status": "reported",
          "value": "mean across three runs per model"
        },
        "system_prompt_public": {
          "notes": "Appendix D.1 publishes the complete prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Only 25 of 41 prompts fit the Llama 3 context limit; the other 16 were counted as insufficient-information responses for accuracy.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One multiple-choice completion per question.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-cloning-scenarios-creator-mcq-llama-context-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 41,
          "notes": "Only 25 prompts fit; 16 were counted as insufficient-information responses for accuracy and coverage.",
          "status": "verified",
          "value": 0.24
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-cloning-scenarios-creator-mcq-llama-context-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 41,
          "notes": "Only 25 prompts fit; 16 were counted as insufficient-information responses for accuracy and coverage.",
          "status": "verified",
          "value": 0.34
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-cloning-scenarios-creator-mcq-llama-context-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 41,
          "notes": "Only 25 prompts fit; 16 were counted as insufficient-information responses for accuracy and coverage.",
          "status": "verified",
          "value": 0.72
        }
      ],
      "scope": {
        "filter": "Full 41-question metric denominator; only 25 prompts fit the model context limit.",
        "n": 41,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Kept separate because the context limit changed realized model calls.",
        "status": "verified"
      },
      "work_id": "lab-bench-paper",
      "work_version_id": "lab-bench-paper-2024-07-14"
    },
    {
      "benchmark_id": "lab-bench-cloning-scenarios",
      "benchmark_version": "paper-v3",
      "comparability_group": "lab-bench-cloning-scenarios-paper-v3-open-response",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-cloning-scenarios-creator-open-response-protocol",
          "locator": {
            "note": "Subset construction, modified wording, expert grading, and second review.",
            "type": "section",
            "value": "Section 2.4 and Appendix B.2"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-cloning-scenarios-creator-open-response-results",
          "locator": {
            "note": "lab-bench-cloning-scenarios model accuracy and question count.",
            "type": "table",
            "value": "Table 5"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-cloning-scenarios-creator-open-response",
      "metrics": [
        {
          "aggregation": "correct / sampled modified questions",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Accuracy",
          "tolerance": "expert judgment against ideal multiple-choice answer",
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "lab-bench-claude-3-5-sonnet-20240620",
        "lab-bench-gpt-4o-unversioned"
      ],
      "protocol": {
        "contamination": {
          "notes": "Main results combine both splits.",
          "reporting_status": "reported",
          "value": "full pre-release snapshot with 80% later public and 20% private contamination-monitoring splits"
        },
        "grader": {
          "human_review": true,
          "model": null,
          "reporting_status": "reported",
          "type": "expert-biologist manual grading against the ideal answer, reviewed by a second expert"
        },
        "reasoning": {
          "notes": "Prompt includes Think step by step.",
          "reporting_status": "reported",
          "value": "zero-shot chain-of-thought"
        },
        "repeats": {
          "notes": "Repeat count is not reported for Table 5.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The paper calls this a small-scale open-answer run but does not restate shot count for the modified prompt.",
          "reporting_status": "not_reported",
          "value": null
        },
        "statistical": {
          "notes": "No confidence intervals reported.",
          "reporting_status": "reported",
          "value": "single reported accuracy per model"
        },
        "system_prompt_public": {
          "notes": "Appendix D.1 publishes the complete prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No common token budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One multiple-choice completion per question.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-cloning-scenarios-creator-open-response-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 10,
          "notes": "Creator Table 5 open-response study.",
          "status": "verified",
          "value": 0.2
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-cloning-scenarios-creator-open-response-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 10,
          "notes": "Creator Table 5 open-response study.",
          "status": "verified",
          "value": 0.2
        }
      ],
      "scope": {
        "filter": "Questions sampled and reworded to remove multiple-choice framing.",
        "n": 10,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "lab-bench-cloning-scenarios-creator-open-response",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Open-response protocol is kept separate from multiple-choice results.",
        "status": "verified"
      },
      "work_id": "lab-bench-paper",
      "work_version_id": "lab-bench-paper-2024-07-14"
    },
    {
      "benchmark_id": "lab-bench-dbqa-dga",
      "benchmark_version": "paper-v3",
      "comparability_group": "lab-bench-dbqa-dga-paper-v3-zero-shot-cot-no-tools",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-dga-creator-mcq-protocol",
          "locator": {
            "note": "Full scope, zero-shot chain-of-thought prompt, no tools, three runs, exact metrics, parser, and model identities.",
            "type": "section",
            "value": "Appendix D.1–D.4; Table 1 and Appendix Table 6"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-dga-creator-mcq-results",
          "locator": {
            "note": "Accuracy, precision, and coverage. Human row excluded.",
            "type": "table",
            "value": "Tables 2–4, row DbQA_dga_task-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-dbqa-dga-creator-mcq",
      "metrics": [
        {
          "aggregation": "correct / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Accuracy",
          "tolerance": "exact choice",
          "unit": "proportion"
        },
        {
          "aggregation": "correct / attempted questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Precision (selective accuracy)",
          "tolerance": "insufficient-information responses excluded from denominator",
          "unit": "proportion"
        },
        {
          "aggregation": "attempted / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "coverage",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Coverage",
          "tolerance": "insufficient-information responses treated as not attempted",
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "lab-bench-claude-3-5-sonnet-20240620",
        "lab-bench-claude-3-haiku-20240307",
        "lab-bench-claude-3-opus-20240229",
        "lab-bench-gemini-1-5-pro-001",
        "lab-bench-gpt-4-turbo-unversioned",
        "lab-bench-gpt-4o-unversioned",
        "lab-bench-meta-llama-3-70b-instruct"
      ],
      "protocol": {
        "contamination": {
          "notes": "Main results combine both splits.",
          "reporting_status": "reported",
          "value": "full pre-release snapshot with 80% later public and 20% private contamination-monitoring splits"
        },
        "grader": {
          "human_review": false,
          "model": "Claude 2 fallback parser",
          "reporting_status": "reported",
          "type": "regex multiple-choice parser with Claude 2 fallback"
        },
        "reasoning": {
          "notes": "Prompt includes Think step by step.",
          "reporting_status": "reported",
          "value": "zero-shot chain-of-thought"
        },
        "repeats": {
          "notes": "Tables 2–4 average three runs per model.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public zero-shot prompt.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "No numeric confidence intervals are reported in Tables 2–4.",
          "reporting_status": "reported",
          "value": "mean across three runs per model"
        },
        "system_prompt_public": {
          "notes": "Appendix D.1 publishes the complete prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No common token budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One multiple-choice completion per question.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-dga-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-dga-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-dga-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-dga-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-dga-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-dga-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.01
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-dga-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-dga-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-dga-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-dga-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.16
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-dga-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.17
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-dga-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.95
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-dga-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.05
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-dga-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.17
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-dga-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.32
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-dga-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.15
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-dga-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.17
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-dga-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.9
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-dga-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.25
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-dga-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.25
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-dga-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.99
        }
      ],
      "scope": {
        "filter": "Complete creator snapshot, combining public and private splits.",
        "n": 50,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "All registered values were parsed from creator Tables 2–4 and independently checked against the HTML table rows.",
        "status": "verified"
      },
      "work_id": "lab-bench-paper",
      "work_version_id": "lab-bench-paper-2024-07-14"
    },
    {
      "benchmark_id": "lab-bench-dbqa-gene-location",
      "benchmark_version": "paper-v3",
      "comparability_group": "lab-bench-dbqa-gene-location-paper-v3-zero-shot-cot-no-tools",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-gene-location-creator-mcq-protocol",
          "locator": {
            "note": "Full scope, zero-shot chain-of-thought prompt, no tools, three runs, exact metrics, parser, and model identities.",
            "type": "section",
            "value": "Appendix D.1–D.4; Table 1 and Appendix Table 6"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-gene-location-creator-mcq-results",
          "locator": {
            "note": "Accuracy, precision, and coverage. Human row excluded.",
            "type": "table",
            "value": "Tables 2–4, row DbQA_gene_location_task-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-dbqa-gene-location-creator-mcq",
      "metrics": [
        {
          "aggregation": "correct / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Accuracy",
          "tolerance": "exact choice",
          "unit": "proportion"
        },
        {
          "aggregation": "correct / attempted questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Precision (selective accuracy)",
          "tolerance": "insufficient-information responses excluded from denominator",
          "unit": "proportion"
        },
        {
          "aggregation": "attempted / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "coverage",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Coverage",
          "tolerance": "insufficient-information responses treated as not attempted",
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "lab-bench-claude-3-5-sonnet-20240620",
        "lab-bench-claude-3-haiku-20240307",
        "lab-bench-claude-3-opus-20240229",
        "lab-bench-gemini-1-5-pro-001",
        "lab-bench-gpt-4-turbo-unversioned",
        "lab-bench-gpt-4o-unversioned",
        "lab-bench-meta-llama-3-70b-instruct"
      ],
      "protocol": {
        "contamination": {
          "notes": "Main results combine both splits.",
          "reporting_status": "reported",
          "value": "full pre-release snapshot with 80% later public and 20% private contamination-monitoring splits"
        },
        "grader": {
          "human_review": false,
          "model": "Claude 2 fallback parser",
          "reporting_status": "reported",
          "type": "regex multiple-choice parser with Claude 2 fallback"
        },
        "reasoning": {
          "notes": "Prompt includes Think step by step.",
          "reporting_status": "reported",
          "value": "zero-shot chain-of-thought"
        },
        "repeats": {
          "notes": "Tables 2–4 average three runs per model.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public zero-shot prompt.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "No numeric confidence intervals are reported in Tables 2–4.",
          "reporting_status": "reported",
          "value": "mean across three runs per model"
        },
        "system_prompt_public": {
          "notes": "Appendix D.1 publishes the complete prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No common token budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One multiple-choice completion per question.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-gene-location-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-gene-location-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-gene-location-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-gene-location-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-gene-location-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-gene-location-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.01
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-gene-location-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.03
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-gene-location-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.22
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-gene-location-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.09
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-gene-location-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.31
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-gene-location-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.38
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-gene-location-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.81
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-gene-location-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.12
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-gene-location-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.22
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-gene-location-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.55
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-gene-location-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.18
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-gene-location-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.22
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-gene-location-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.81
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-gene-location-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.2
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-gene-location-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.21
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-gene-location-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.97
        }
      ],
      "scope": {
        "filter": "Complete creator snapshot, combining public and private splits.",
        "n": 50,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "All registered values were parsed from creator Tables 2–4 and independently checked against the HTML table rows.",
        "status": "verified"
      },
      "work_id": "lab-bench-paper",
      "work_version_id": "lab-bench-paper-2024-07-14"
    },
    {
      "benchmark_id": "lab-bench-dbqa-mirna-targets",
      "benchmark_version": "paper-v3",
      "comparability_group": "lab-bench-dbqa-mirna-targets-paper-v3-zero-shot-cot-no-tools",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-mirna-targets-creator-mcq-protocol",
          "locator": {
            "note": "Full scope, zero-shot chain-of-thought prompt, no tools, three runs, exact metrics, parser, and model identities.",
            "type": "section",
            "value": "Appendix D.1–D.4; Table 1 and Appendix Table 6"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-mirna-targets-creator-mcq-results",
          "locator": {
            "note": "Accuracy, precision, and coverage. Human row excluded.",
            "type": "table",
            "value": "Tables 2–4, row DbQA_mirna_targets_task-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-dbqa-mirna-targets-creator-mcq",
      "metrics": [
        {
          "aggregation": "correct / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Accuracy",
          "tolerance": "exact choice",
          "unit": "proportion"
        },
        {
          "aggregation": "correct / attempted questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Precision (selective accuracy)",
          "tolerance": "insufficient-information responses excluded from denominator",
          "unit": "proportion"
        },
        {
          "aggregation": "attempted / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "coverage",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Coverage",
          "tolerance": "insufficient-information responses treated as not attempted",
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "lab-bench-claude-3-5-sonnet-20240620",
        "lab-bench-claude-3-haiku-20240307",
        "lab-bench-claude-3-opus-20240229",
        "lab-bench-gemini-1-5-pro-001",
        "lab-bench-gpt-4-turbo-unversioned",
        "lab-bench-gpt-4o-unversioned",
        "lab-bench-meta-llama-3-70b-instruct"
      ],
      "protocol": {
        "contamination": {
          "notes": "Main results combine both splits.",
          "reporting_status": "reported",
          "value": "full pre-release snapshot with 80% later public and 20% private contamination-monitoring splits"
        },
        "grader": {
          "human_review": false,
          "model": "Claude 2 fallback parser",
          "reporting_status": "reported",
          "type": "regex multiple-choice parser with Claude 2 fallback"
        },
        "reasoning": {
          "notes": "Prompt includes Think step by step.",
          "reporting_status": "reported",
          "value": "zero-shot chain-of-thought"
        },
        "repeats": {
          "notes": "Tables 2–4 average three runs per model.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public zero-shot prompt.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "No numeric confidence intervals are reported in Tables 2–4.",
          "reporting_status": "reported",
          "value": "mean across three runs per model"
        },
        "system_prompt_public": {
          "notes": "Appendix D.1 publishes the complete prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No common token budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One multiple-choice completion per question.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mirna-targets-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mirna-targets-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mirna-targets-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mirna-targets-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mirna-targets-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mirna-targets-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mirna-targets-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mirna-targets-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mirna-targets-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mirna-targets-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.1
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mirna-targets-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.3
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mirna-targets-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.35
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mirna-targets-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.01
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mirna-targets-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.03
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mirna-targets-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.15
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mirna-targets-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.15
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mirna-targets-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.26
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mirna-targets-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.6
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mirna-targets-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.29
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mirna-targets-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.29
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mirna-targets-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.99
        }
      ],
      "scope": {
        "filter": "Complete creator snapshot, combining public and private splits.",
        "n": 50,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "All registered values were parsed from creator Tables 2–4 and independently checked against the HTML table rows.",
        "status": "verified"
      },
      "work_id": "lab-bench-paper",
      "work_version_id": "lab-bench-paper-2024-07-14"
    },
    {
      "benchmark_id": "lab-bench-dbqa-mouse-tumor-gene-sets",
      "benchmark_version": "paper-v3",
      "comparability_group": "lab-bench-dbqa-mouse-tumor-gene-sets-paper-v3-zero-shot-cot-no-tools",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-mouse-tumor-gene-sets-creator-mcq-protocol",
          "locator": {
            "note": "Full scope, zero-shot chain-of-thought prompt, no tools, three runs, exact metrics, parser, and model identities.",
            "type": "section",
            "value": "Appendix D.1–D.4; Table 1 and Appendix Table 6"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-mouse-tumor-gene-sets-creator-mcq-results",
          "locator": {
            "note": "Accuracy, precision, and coverage. Human row excluded.",
            "type": "table",
            "value": "Tables 2–4, row DbQA_mouse_tumor_gene_sets-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-dbqa-mouse-tumor-gene-sets-creator-mcq",
      "metrics": [
        {
          "aggregation": "correct / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Accuracy",
          "tolerance": "exact choice",
          "unit": "proportion"
        },
        {
          "aggregation": "correct / attempted questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Precision (selective accuracy)",
          "tolerance": "insufficient-information responses excluded from denominator",
          "unit": "proportion"
        },
        {
          "aggregation": "attempted / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "coverage",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Coverage",
          "tolerance": "insufficient-information responses treated as not attempted",
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "lab-bench-claude-3-5-sonnet-20240620",
        "lab-bench-claude-3-haiku-20240307",
        "lab-bench-claude-3-opus-20240229",
        "lab-bench-gemini-1-5-pro-001",
        "lab-bench-gpt-4-turbo-unversioned",
        "lab-bench-gpt-4o-unversioned",
        "lab-bench-meta-llama-3-70b-instruct"
      ],
      "protocol": {
        "contamination": {
          "notes": "Main results combine both splits.",
          "reporting_status": "reported",
          "value": "full pre-release snapshot with 80% later public and 20% private contamination-monitoring splits"
        },
        "grader": {
          "human_review": false,
          "model": "Claude 2 fallback parser",
          "reporting_status": "reported",
          "type": "regex multiple-choice parser with Claude 2 fallback"
        },
        "reasoning": {
          "notes": "Prompt includes Think step by step.",
          "reporting_status": "reported",
          "value": "zero-shot chain-of-thought"
        },
        "repeats": {
          "notes": "Tables 2–4 average three runs per model.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public zero-shot prompt.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "No numeric confidence intervals are reported in Tables 2–4.",
          "reporting_status": "reported",
          "value": "mean across three runs per model"
        },
        "system_prompt_public": {
          "notes": "Appendix D.1 publishes the complete prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No common token budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One multiple-choice completion per question.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mouse-tumor-gene-sets-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.55
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mouse-tumor-gene-sets-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.73
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mouse-tumor-gene-sets-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.76
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mouse-tumor-gene-sets-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.36
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mouse-tumor-gene-sets-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.71
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mouse-tumor-gene-sets-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.51
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mouse-tumor-gene-sets-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.1
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mouse-tumor-gene-sets-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.76
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mouse-tumor-gene-sets-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.14
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mouse-tumor-gene-sets-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.57
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mouse-tumor-gene-sets-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.59
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mouse-tumor-gene-sets-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.96
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mouse-tumor-gene-sets-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.16
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mouse-tumor-gene-sets-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.58
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mouse-tumor-gene-sets-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.28
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mouse-tumor-gene-sets-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.48
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mouse-tumor-gene-sets-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.55
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mouse-tumor-gene-sets-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.87
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mouse-tumor-gene-sets-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.53
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mouse-tumor-gene-sets-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.58
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-mouse-tumor-gene-sets-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.93
        }
      ],
      "scope": {
        "filter": "Complete creator snapshot, combining public and private splits.",
        "n": 100,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "All registered values were parsed from creator Tables 2–4 and independently checked against the HTML table rows.",
        "status": "verified"
      },
      "work_id": "lab-bench-paper",
      "work_version_id": "lab-bench-paper-2024-07-14"
    },
    {
      "benchmark_id": "lab-bench-dbqa-oncogenic-signatures",
      "benchmark_version": "paper-v3",
      "comparability_group": "lab-bench-dbqa-oncogenic-signatures-paper-v3-zero-shot-cot-no-tools",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-oncogenic-signatures-creator-mcq-protocol",
          "locator": {
            "note": "Full scope, zero-shot chain-of-thought prompt, no tools, three runs, exact metrics, parser, and model identities.",
            "type": "section",
            "value": "Appendix D.1–D.4; Table 1 and Appendix Table 6"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-oncogenic-signatures-creator-mcq-results",
          "locator": {
            "note": "Accuracy, precision, and coverage. Human row excluded.",
            "type": "table",
            "value": "Tables 2–4, row DbQA_oncogenic_signatures_task-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-dbqa-oncogenic-signatures-creator-mcq",
      "metrics": [
        {
          "aggregation": "correct / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Accuracy",
          "tolerance": "exact choice",
          "unit": "proportion"
        },
        {
          "aggregation": "correct / attempted questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Precision (selective accuracy)",
          "tolerance": "insufficient-information responses excluded from denominator",
          "unit": "proportion"
        },
        {
          "aggregation": "attempted / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "coverage",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Coverage",
          "tolerance": "insufficient-information responses treated as not attempted",
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "lab-bench-claude-3-5-sonnet-20240620",
        "lab-bench-claude-3-haiku-20240307",
        "lab-bench-claude-3-opus-20240229",
        "lab-bench-gemini-1-5-pro-001",
        "lab-bench-gpt-4-turbo-unversioned",
        "lab-bench-gpt-4o-unversioned",
        "lab-bench-meta-llama-3-70b-instruct"
      ],
      "protocol": {
        "contamination": {
          "notes": "Main results combine both splits.",
          "reporting_status": "reported",
          "value": "full pre-release snapshot with 80% later public and 20% private contamination-monitoring splits"
        },
        "grader": {
          "human_review": false,
          "model": "Claude 2 fallback parser",
          "reporting_status": "reported",
          "type": "regex multiple-choice parser with Claude 2 fallback"
        },
        "reasoning": {
          "notes": "Prompt includes Think step by step.",
          "reporting_status": "reported",
          "value": "zero-shot chain-of-thought"
        },
        "repeats": {
          "notes": "Tables 2–4 average three runs per model.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public zero-shot prompt.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "No numeric confidence intervals are reported in Tables 2–4.",
          "reporting_status": "reported",
          "value": "mean across three runs per model"
        },
        "system_prompt_public": {
          "notes": "Appendix D.1 publishes the complete prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No common token budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One multiple-choice completion per question.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-oncogenic-signatures-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.14
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-oncogenic-signatures-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.59
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-oncogenic-signatures-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.24
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-oncogenic-signatures-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.12
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-oncogenic-signatures-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.57
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-oncogenic-signatures-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.2
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-oncogenic-signatures-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.04
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-oncogenic-signatures-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.75
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-oncogenic-signatures-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.05
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-oncogenic-signatures-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.25
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-oncogenic-signatures-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.32
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-oncogenic-signatures-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.79
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-oncogenic-signatures-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.06
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-oncogenic-signatures-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.61
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-oncogenic-signatures-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.1
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-oncogenic-signatures-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.24
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-oncogenic-signatures-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.3
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-oncogenic-signatures-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.8
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-oncogenic-signatures-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.2
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-oncogenic-signatures-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.47
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-oncogenic-signatures-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.42
        }
      ],
      "scope": {
        "filter": "Complete creator snapshot, combining public and private splits.",
        "n": 50,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "All registered values were parsed from creator Tables 2–4 and independently checked against the HTML table rows.",
        "status": "verified"
      },
      "work_id": "lab-bench-paper",
      "work_version_id": "lab-bench-paper-2024-07-14"
    },
    {
      "benchmark_id": "lab-bench-dbqa-tfbs-gtrd",
      "benchmark_version": "paper-v3",
      "comparability_group": "lab-bench-dbqa-tfbs-gtrd-paper-v3-zero-shot-cot-no-tools",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-tfbs-gtrd-creator-mcq-protocol",
          "locator": {
            "note": "Full scope, zero-shot chain-of-thought prompt, no tools, three runs, exact metrics, parser, and model identities.",
            "type": "section",
            "value": "Appendix D.1–D.4; Table 1 and Appendix Table 6"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-tfbs-gtrd-creator-mcq-results",
          "locator": {
            "note": "Accuracy, precision, and coverage. Human row excluded.",
            "type": "table",
            "value": "Tables 2–4, row DbQA_tfbs_GTRD_task-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-dbqa-tfbs-gtrd-creator-mcq",
      "metrics": [
        {
          "aggregation": "correct / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Accuracy",
          "tolerance": "exact choice",
          "unit": "proportion"
        },
        {
          "aggregation": "correct / attempted questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Precision (selective accuracy)",
          "tolerance": "insufficient-information responses excluded from denominator",
          "unit": "proportion"
        },
        {
          "aggregation": "attempted / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "coverage",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Coverage",
          "tolerance": "insufficient-information responses treated as not attempted",
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "lab-bench-claude-3-5-sonnet-20240620",
        "lab-bench-claude-3-haiku-20240307",
        "lab-bench-claude-3-opus-20240229",
        "lab-bench-gemini-1-5-pro-001",
        "lab-bench-gpt-4-turbo-unversioned",
        "lab-bench-gpt-4o-unversioned",
        "lab-bench-meta-llama-3-70b-instruct"
      ],
      "protocol": {
        "contamination": {
          "notes": "Main results combine both splits.",
          "reporting_status": "reported",
          "value": "full pre-release snapshot with 80% later public and 20% private contamination-monitoring splits"
        },
        "grader": {
          "human_review": false,
          "model": "Claude 2 fallback parser",
          "reporting_status": "reported",
          "type": "regex multiple-choice parser with Claude 2 fallback"
        },
        "reasoning": {
          "notes": "Prompt includes Think step by step.",
          "reporting_status": "reported",
          "value": "zero-shot chain-of-thought"
        },
        "repeats": {
          "notes": "Tables 2–4 average three runs per model.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public zero-shot prompt.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "No numeric confidence intervals are reported in Tables 2–4.",
          "reporting_status": "reported",
          "value": "mean across three runs per model"
        },
        "system_prompt_public": {
          "notes": "Appendix D.1 publishes the complete prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No common token budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One multiple-choice completion per question.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-tfbs-gtrd-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-tfbs-gtrd-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-tfbs-gtrd-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-tfbs-gtrd-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-tfbs-gtrd-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-tfbs-gtrd-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-tfbs-gtrd-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-tfbs-gtrd-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-tfbs-gtrd-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.01
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-tfbs-gtrd-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.01
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-tfbs-gtrd-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.33
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-tfbs-gtrd-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.01
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-tfbs-gtrd-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-tfbs-gtrd-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-tfbs-gtrd-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.11
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-tfbs-gtrd-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.11
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-tfbs-gtrd-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.31
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-tfbs-gtrd-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.35
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-tfbs-gtrd-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.19
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-tfbs-gtrd-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.19
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-tfbs-gtrd-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 1.0
        }
      ],
      "scope": {
        "filter": "Complete creator snapshot, combining public and private splits.",
        "n": 50,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "All registered values were parsed from creator Tables 2–4 and independently checked against the HTML table rows.",
        "status": "verified"
      },
      "work_id": "lab-bench-paper",
      "work_version_id": "lab-bench-paper-2024-07-14"
    },
    {
      "benchmark_id": "lab-bench-dbqa-variant-from-sequence",
      "benchmark_version": "paper-v3",
      "comparability_group": "lab-bench-dbqa-variant-from-sequence-paper-v3-zero-shot-cot-no-tools",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-variant-from-sequence-creator-mcq-protocol",
          "locator": {
            "note": "Full scope, zero-shot chain-of-thought prompt, no tools, three runs, exact metrics, parser, and model identities.",
            "type": "section",
            "value": "Appendix D.1–D.4; Table 1 and Appendix Table 6"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-variant-from-sequence-creator-mcq-results",
          "locator": {
            "note": "Accuracy, precision, and coverage. Human row excluded.",
            "type": "table",
            "value": "Tables 2–4, row DbQA_variant_from_sequence_task-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-dbqa-variant-from-sequence-creator-mcq",
      "metrics": [
        {
          "aggregation": "correct / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Accuracy",
          "tolerance": "exact choice",
          "unit": "proportion"
        },
        {
          "aggregation": "correct / attempted questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Precision (selective accuracy)",
          "tolerance": "insufficient-information responses excluded from denominator",
          "unit": "proportion"
        },
        {
          "aggregation": "attempted / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "coverage",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Coverage",
          "tolerance": "insufficient-information responses treated as not attempted",
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "lab-bench-claude-3-5-sonnet-20240620",
        "lab-bench-claude-3-haiku-20240307",
        "lab-bench-claude-3-opus-20240229",
        "lab-bench-gemini-1-5-pro-001",
        "lab-bench-gpt-4-turbo-unversioned",
        "lab-bench-gpt-4o-unversioned",
        "lab-bench-meta-llama-3-70b-instruct"
      ],
      "protocol": {
        "contamination": {
          "notes": "Main results combine both splits.",
          "reporting_status": "reported",
          "value": "full pre-release snapshot with 80% later public and 20% private contamination-monitoring splits"
        },
        "grader": {
          "human_review": false,
          "model": "Claude 2 fallback parser",
          "reporting_status": "reported",
          "type": "regex multiple-choice parser with Claude 2 fallback"
        },
        "reasoning": {
          "notes": "Prompt includes Think step by step.",
          "reporting_status": "reported",
          "value": "zero-shot chain-of-thought"
        },
        "repeats": {
          "notes": "Tables 2–4 average three runs per model.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public zero-shot prompt.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "No numeric confidence intervals are reported in Tables 2–4.",
          "reporting_status": "reported",
          "value": "mean across three runs per model"
        },
        "system_prompt_public": {
          "notes": "Appendix D.1 publishes the complete prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No common token budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One multiple-choice completion per question.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-from-sequence-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.02
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-from-sequence-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.75
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-from-sequence-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.03
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-from-sequence-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-from-sequence-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.11
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-from-sequence-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.03
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-from-sequence-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-from-sequence-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.17
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-from-sequence-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.01
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-from-sequence-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.42
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-from-sequence-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.44
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-from-sequence-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.95
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-from-sequence-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-from-sequence-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-from-sequence-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.01
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-from-sequence-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.22
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-from-sequence-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.36
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-from-sequence-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.62
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-from-sequence-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.3
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-from-sequence-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.35
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-from-sequence-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.85
        }
      ],
      "scope": {
        "filter": "Complete creator snapshot, combining public and private splits.",
        "n": 100,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "All registered values were parsed from creator Tables 2–4 and independently checked against the HTML table rows.",
        "status": "verified"
      },
      "work_id": "lab-bench-paper",
      "work_version_id": "lab-bench-paper-2024-07-14"
    },
    {
      "benchmark_id": "lab-bench-dbqa-variant-multi-sequence",
      "benchmark_version": "paper-v3",
      "comparability_group": "lab-bench-dbqa-variant-multi-sequence-paper-v3-zero-shot-cot-no-tools",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-variant-multi-sequence-creator-mcq-protocol",
          "locator": {
            "note": "Full scope, zero-shot chain-of-thought prompt, no tools, three runs, exact metrics, parser, and model identities.",
            "type": "section",
            "value": "Appendix D.1–D.4; Table 1 and Appendix Table 6"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-variant-multi-sequence-creator-mcq-results",
          "locator": {
            "note": "Accuracy, precision, and coverage. Human row excluded.",
            "type": "table",
            "value": "Tables 2–4, row DbQA_variant_multi_sequence_task-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-dbqa-variant-multi-sequence-creator-mcq",
      "metrics": [
        {
          "aggregation": "correct / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Accuracy",
          "tolerance": "exact choice",
          "unit": "proportion"
        },
        {
          "aggregation": "correct / attempted questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Precision (selective accuracy)",
          "tolerance": "insufficient-information responses excluded from denominator",
          "unit": "proportion"
        },
        {
          "aggregation": "attempted / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "coverage",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Coverage",
          "tolerance": "insufficient-information responses treated as not attempted",
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "lab-bench-claude-3-5-sonnet-20240620",
        "lab-bench-claude-3-haiku-20240307",
        "lab-bench-claude-3-opus-20240229",
        "lab-bench-gemini-1-5-pro-001",
        "lab-bench-gpt-4-turbo-unversioned",
        "lab-bench-gpt-4o-unversioned",
        "lab-bench-meta-llama-3-70b-instruct"
      ],
      "protocol": {
        "contamination": {
          "notes": "Main results combine both splits.",
          "reporting_status": "reported",
          "value": "full pre-release snapshot with 80% later public and 20% private contamination-monitoring splits"
        },
        "grader": {
          "human_review": false,
          "model": "Claude 2 fallback parser",
          "reporting_status": "reported",
          "type": "regex multiple-choice parser with Claude 2 fallback"
        },
        "reasoning": {
          "notes": "Prompt includes Think step by step.",
          "reporting_status": "reported",
          "value": "zero-shot chain-of-thought"
        },
        "repeats": {
          "notes": "Tables 2–4 average three runs per model.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public zero-shot prompt.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "No numeric confidence intervals are reported in Tables 2–4.",
          "reporting_status": "reported",
          "value": "mean across three runs per model"
        },
        "system_prompt_public": {
          "notes": "Appendix D.1 publishes the complete prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No common token budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One multiple-choice completion per question.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-multi-sequence-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.09
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-multi-sequence-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.16
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-multi-sequence-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.56
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-multi-sequence-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.05
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-multi-sequence-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.18
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-multi-sequence-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.26
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-multi-sequence-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-multi-sequence-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.33
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-multi-sequence-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.04
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-multi-sequence-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.06
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-multi-sequence-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.12
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-multi-sequence-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.49
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-multi-sequence-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-multi-sequence-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-multi-sequence-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.02
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-multi-sequence-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.14
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-multi-sequence-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.16
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-multi-sequence-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.84
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-multi-sequence-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.07
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-multi-sequence-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.12
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-variant-multi-sequence-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 100,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.63
        }
      ],
      "scope": {
        "filter": "Complete creator snapshot, combining public and private splits.",
        "n": 100,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "All registered values were parsed from creator Tables 2–4 and independently checked against the HTML table rows.",
        "status": "verified"
      },
      "work_id": "lab-bench-paper",
      "work_version_id": "lab-bench-paper-2024-07-14"
    },
    {
      "benchmark_id": "lab-bench-dbqa-vax-response",
      "benchmark_version": "paper-v3",
      "comparability_group": "lab-bench-dbqa-vax-response-paper-v3-zero-shot-cot-no-tools",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-vax-response-creator-mcq-protocol",
          "locator": {
            "note": "Full scope, zero-shot chain-of-thought prompt, no tools, three runs, exact metrics, parser, and model identities.",
            "type": "section",
            "value": "Appendix D.1–D.4; Table 1 and Appendix Table 6"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-vax-response-creator-mcq-results",
          "locator": {
            "note": "Accuracy, precision, and coverage. Human row excluded.",
            "type": "table",
            "value": "Tables 2–4, row DbQA_vax_response_task-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-dbqa-vax-response-creator-mcq",
      "metrics": [
        {
          "aggregation": "correct / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Accuracy",
          "tolerance": "exact choice",
          "unit": "proportion"
        },
        {
          "aggregation": "correct / attempted questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Precision (selective accuracy)",
          "tolerance": "insufficient-information responses excluded from denominator",
          "unit": "proportion"
        },
        {
          "aggregation": "attempted / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "coverage",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Coverage",
          "tolerance": "insufficient-information responses treated as not attempted",
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "lab-bench-claude-3-5-sonnet-20240620",
        "lab-bench-claude-3-haiku-20240307",
        "lab-bench-claude-3-opus-20240229",
        "lab-bench-gemini-1-5-pro-001",
        "lab-bench-gpt-4-turbo-unversioned",
        "lab-bench-gpt-4o-unversioned",
        "lab-bench-meta-llama-3-70b-instruct"
      ],
      "protocol": {
        "contamination": {
          "notes": "Main results combine both splits.",
          "reporting_status": "reported",
          "value": "full pre-release snapshot with 80% later public and 20% private contamination-monitoring splits"
        },
        "grader": {
          "human_review": false,
          "model": "Claude 2 fallback parser",
          "reporting_status": "reported",
          "type": "regex multiple-choice parser with Claude 2 fallback"
        },
        "reasoning": {
          "notes": "Prompt includes Think step by step.",
          "reporting_status": "reported",
          "value": "zero-shot chain-of-thought"
        },
        "repeats": {
          "notes": "Tables 2–4 average three runs per model.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public zero-shot prompt.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "No numeric confidence intervals are reported in Tables 2–4.",
          "reporting_status": "reported",
          "value": "mean across three runs per model"
        },
        "system_prompt_public": {
          "notes": "Appendix D.1 publishes the complete prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No common token budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One multiple-choice completion per question.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-vax-response-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.21
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-vax-response-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.57
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-vax-response-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.36
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-vax-response-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.14
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-vax-response-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.65
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-vax-response-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.21
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-vax-response-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.13
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-vax-response-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.61
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-vax-response-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.21
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-vax-response-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.33
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-vax-response-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.45
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-vax-response-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.73
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-vax-response-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.06
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-vax-response-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.65
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-vax-response-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.09
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-vax-response-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.32
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-vax-response-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.38
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-vax-response-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.83
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-vax-response-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.15
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-vax-response-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.54
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-vax-response-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.29
        }
      ],
      "scope": {
        "filter": "Complete creator snapshot, combining public and private splits.",
        "n": 50,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "All registered values were parsed from creator Tables 2–4 and independently checked against the HTML table rows.",
        "status": "verified"
      },
      "work_id": "lab-bench-paper",
      "work_version_id": "lab-bench-paper-2024-07-14"
    },
    {
      "benchmark_id": "lab-bench-dbqa-viral-ppi",
      "benchmark_version": "paper-v3",
      "comparability_group": "lab-bench-dbqa-viral-ppi-paper-v3-zero-shot-cot-no-tools",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-viral-ppi-creator-mcq-protocol",
          "locator": {
            "note": "Full scope, zero-shot chain-of-thought prompt, no tools, three runs, exact metrics, parser, and model identities.",
            "type": "section",
            "value": "Appendix D.1–D.4; Table 1 and Appendix Table 6"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-dbqa-viral-ppi-creator-mcq-results",
          "locator": {
            "note": "Accuracy, precision, and coverage. Human row excluded.",
            "type": "table",
            "value": "Tables 2–4, row DbQA_viral_ppi_task-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-dbqa-viral-ppi-creator-mcq",
      "metrics": [
        {
          "aggregation": "correct / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Accuracy",
          "tolerance": "exact choice",
          "unit": "proportion"
        },
        {
          "aggregation": "correct / attempted questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Precision (selective accuracy)",
          "tolerance": "insufficient-information responses excluded from denominator",
          "unit": "proportion"
        },
        {
          "aggregation": "attempted / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "coverage",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Coverage",
          "tolerance": "insufficient-information responses treated as not attempted",
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "lab-bench-claude-3-5-sonnet-20240620",
        "lab-bench-claude-3-haiku-20240307",
        "lab-bench-claude-3-opus-20240229",
        "lab-bench-gemini-1-5-pro-001",
        "lab-bench-gpt-4-turbo-unversioned",
        "lab-bench-gpt-4o-unversioned",
        "lab-bench-meta-llama-3-70b-instruct"
      ],
      "protocol": {
        "contamination": {
          "notes": "Main results combine both splits.",
          "reporting_status": "reported",
          "value": "full pre-release snapshot with 80% later public and 20% private contamination-monitoring splits"
        },
        "grader": {
          "human_review": false,
          "model": "Claude 2 fallback parser",
          "reporting_status": "reported",
          "type": "regex multiple-choice parser with Claude 2 fallback"
        },
        "reasoning": {
          "notes": "Prompt includes Think step by step.",
          "reporting_status": "reported",
          "value": "zero-shot chain-of-thought"
        },
        "repeats": {
          "notes": "Tables 2–4 average three runs per model.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public zero-shot prompt.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "No numeric confidence intervals are reported in Tables 2–4.",
          "reporting_status": "reported",
          "value": "mean across three runs per model"
        },
        "system_prompt_public": {
          "notes": "Appendix D.1 publishes the complete prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No common token budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One multiple-choice completion per question.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-viral-ppi-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-viral-ppi-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-viral-ppi-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-viral-ppi-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-viral-ppi-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-viral-ppi-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-viral-ppi-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-viral-ppi-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-viral-ppi-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-viral-ppi-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.26
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-viral-ppi-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.69
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-viral-ppi-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.38
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-viral-ppi-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.02
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-viral-ppi-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.28
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-viral-ppi-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.06
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-viral-ppi-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.27
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-viral-ppi-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.49
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-viral-ppi-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.55
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-viral-ppi-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.4
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-viral-ppi-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.4
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-dbqa-viral-ppi-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 1.0
        }
      ],
      "scope": {
        "filter": "Complete creator snapshot, combining public and private splits.",
        "n": 50,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "All registered values were parsed from creator Tables 2–4 and independently checked against the HTML table rows.",
        "status": "verified"
      },
      "work_id": "lab-bench-paper",
      "work_version_id": "lab-bench-paper-2024-07-14"
    },
    {
      "benchmark_id": "lab-bench-figqa",
      "benchmark_version": null,
      "comparability_group": "lab-bench-figqa-anthropic-10shot-no-tools",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-figqa-anthropic-sonnet45-system-card-protocol",
          "locator": {
            "note": "Four selected tracks, 10-shot prompting, no search/bioinformatics tools, threshold, and unreported scope/repeats/grader.",
            "type": "section",
            "value": "Section 9.2.4.4, printed pages 132–133"
          },
          "source_id": "anthropic-sonnet-4-5-system-card",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-figqa-anthropic-sonnet45-system-card-results",
          "locator": {
            "note": "Exact point labels. Protocol Sonnet 4/4.5 labels are represented in the Claude for Life Sciences run to avoid duplicate result rows.",
            "type": "figure",
            "value": "Figure 9.2.4.4.A, printed page 133"
          },
          "source_id": "anthropic-sonnet-4-5-system-card",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-figqa-anthropic-sonnet45-system-card",
      "metrics": [
        {
          "aggregation": null,
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "LAB-Bench score",
          "tolerance": null,
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "claude-opus-4",
        "claude-opus-4-1",
        "claude-sonnet-4",
        "claude-sonnet-4-5"
      ],
      "protocol": {
        "contamination": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": null,
          "model": null,
          "reporting_status": "not_reported",
          "type": null
        },
        "reasoning": {
          "notes": "Thinking/effort configuration is not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "repeats": {
          "notes": "The number of repeats underlying the plotted error bars is not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Figure title and caption specify 10-shot prompting.",
          "reporting_status": "reported",
          "value": 10
        },
        "statistical": {
          "notes": "The figure does not give numeric interval bounds.",
          "reporting_status": "reported",
          "value": "reported point estimate with plotted error bars"
        },
        "system_prompt_public": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "The system card says search and bioinformatics tools were not included in the testing environment.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "The system card says search and bioinformatics tools were not included in the testing environment.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "The system card says search and bioinformatics tools were not included in the testing environment.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "The system card says search and bioinformatics tools were not included in the testing environment.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "The system card says search and bioinformatics tools were not included in the testing environment.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "The system card says search and bioinformatics tools were not included in the testing environment.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "Multiple-choice evaluation.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-figqa-anthropic-sonnet45-system-card-results"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-sonnet-4",
          "n": null,
          "notes": "Point label transcribed from Figure 9.2.4.4.A; numeric error-bar bounds are not reported.",
          "status": "verified",
          "value": 0.398
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-figqa-anthropic-sonnet45-system-card-results"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-opus-4",
          "n": null,
          "notes": "Point label transcribed from Figure 9.2.4.4.A; numeric error-bar bounds are not reported.",
          "status": "verified",
          "value": 0.508
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-figqa-anthropic-sonnet45-system-card-results"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-opus-4-1",
          "n": null,
          "notes": "Point label transcribed from Figure 9.2.4.4.A; numeric error-bar bounds are not reported.",
          "status": "verified",
          "value": 0.481
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-figqa-anthropic-sonnet45-system-card-results"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-sonnet-4-5",
          "n": null,
          "notes": "Point label transcribed from Figure 9.2.4.4.A; numeric error-bar bounds are not reported.",
          "status": "verified",
          "value": 0.497
        }
      ],
      "scope": {
        "filter": "lab-bench-figqa track; benchmark snapshot, public/private scope, and realized n are not reported.",
        "n": null,
        "reporting_status": "not_reported",
        "selection": null,
        "subset_id": null,
        "type": "track"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Protocol, exact point labels, and missing n/repeat details were checked on printed pages 132–133.",
        "status": "verified"
      },
      "work_id": "anthropic-sonnet-4-5-system-card",
      "work_version_id": "anthropic-sonnet-4-5-system-card-2025-09-29"
    },
    {
      "benchmark_id": "lab-bench-figqa",
      "benchmark_version": "paper-v3",
      "comparability_group": "lab-bench-figqa-paper-v3-zero-shot-cot-no-tools",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-figqa-creator-mcq-protocol",
          "locator": {
            "note": "Full scope, zero-shot chain-of-thought prompt, no tools, three runs, exact metrics, parser, and model identities.",
            "type": "section",
            "value": "Appendix D.1–D.4; Table 1 and Appendix Table 6"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-figqa-creator-mcq-results",
          "locator": {
            "note": "Accuracy, precision, and coverage. Human row excluded.",
            "type": "table",
            "value": "Tables 2–4, row FigQA"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-figqa-creator-mcq",
      "metrics": [
        {
          "aggregation": "correct / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Accuracy",
          "tolerance": "exact choice",
          "unit": "proportion"
        },
        {
          "aggregation": "correct / attempted questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Precision (selective accuracy)",
          "tolerance": "insufficient-information responses excluded from denominator",
          "unit": "proportion"
        },
        {
          "aggregation": "attempted / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "coverage",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Coverage",
          "tolerance": "insufficient-information responses treated as not attempted",
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "lab-bench-claude-3-5-sonnet-20240620",
        "lab-bench-claude-3-haiku-20240307",
        "lab-bench-claude-3-opus-20240229",
        "lab-bench-gemini-1-5-pro-001",
        "lab-bench-gpt-4-turbo-unversioned",
        "lab-bench-gpt-4o-unversioned"
      ],
      "protocol": {
        "contamination": {
          "notes": "Main results combine both splits.",
          "reporting_status": "reported",
          "value": "full pre-release snapshot with 80% later public and 20% private contamination-monitoring splits"
        },
        "grader": {
          "human_review": false,
          "model": "Claude 2 fallback parser",
          "reporting_status": "reported",
          "type": "regex multiple-choice parser with Claude 2 fallback"
        },
        "reasoning": {
          "notes": "Prompt includes Think step by step.",
          "reporting_status": "reported",
          "value": "zero-shot chain-of-thought"
        },
        "repeats": {
          "notes": "Tables 2–4 average three runs per model.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public zero-shot prompt.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "No numeric confidence intervals are reported in Tables 2–4.",
          "reporting_status": "reported",
          "value": "mean across three runs per model"
        },
        "system_prompt_public": {
          "notes": "Appendix D.1 publishes the complete prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No common token budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One multiple-choice completion per question.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-figqa-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 226,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.46
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-figqa-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 226,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.54
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-figqa-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 226,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.85
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-figqa-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 226,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.24
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-figqa-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 226,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.31
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-figqa-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 226,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.78
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-figqa-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 226,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.25
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-figqa-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 226,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.34
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-figqa-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 226,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.73
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-figqa-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 226,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.29
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-figqa-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 226,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.3
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-figqa-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 226,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.97
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-figqa-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 226,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.25
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-figqa-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 226,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.28
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-figqa-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 226,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.91
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-figqa-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 226,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.23
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-figqa-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 226,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.24
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-figqa-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 226,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.95
        }
      ],
      "scope": {
        "filter": "Complete creator snapshot, combining public and private splits.",
        "n": 226,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "All registered values were parsed from creator Tables 2–4 and independently checked against the HTML table rows.",
        "status": "verified"
      },
      "work_id": "lab-bench-paper",
      "work_version_id": "lab-bench-paper-2024-07-14"
    },
    {
      "benchmark_id": "lab-bench-figqa",
      "benchmark_version": "paper-v3",
      "comparability_group": "lab-bench-figqa-paper-v3-open-response",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-figqa-creator-open-response-protocol",
          "locator": {
            "note": "Subset construction, modified wording, expert grading, and second review.",
            "type": "section",
            "value": "Section 2.4 and Appendix B.2"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-figqa-creator-open-response-results",
          "locator": {
            "note": "lab-bench-figqa model accuracy and question count.",
            "type": "table",
            "value": "Table 5"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-figqa-creator-open-response",
      "metrics": [
        {
          "aggregation": "correct / sampled modified questions",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Accuracy",
          "tolerance": "expert judgment against ideal multiple-choice answer",
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "lab-bench-claude-3-5-sonnet-20240620",
        "lab-bench-gpt-4o-unversioned"
      ],
      "protocol": {
        "contamination": {
          "notes": "Main results combine both splits.",
          "reporting_status": "reported",
          "value": "full pre-release snapshot with 80% later public and 20% private contamination-monitoring splits"
        },
        "grader": {
          "human_review": true,
          "model": null,
          "reporting_status": "reported",
          "type": "expert-biologist manual grading against the ideal answer, reviewed by a second expert"
        },
        "reasoning": {
          "notes": "Prompt includes Think step by step.",
          "reporting_status": "reported",
          "value": "zero-shot chain-of-thought"
        },
        "repeats": {
          "notes": "Repeat count is not reported for Table 5.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The paper calls this a small-scale open-answer run but does not restate shot count for the modified prompt.",
          "reporting_status": "not_reported",
          "value": null
        },
        "statistical": {
          "notes": "No confidence intervals reported.",
          "reporting_status": "reported",
          "value": "single reported accuracy per model"
        },
        "system_prompt_public": {
          "notes": "Appendix D.1 publishes the complete prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No common token budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One multiple-choice completion per question.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-figqa-creator-open-response-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 10,
          "notes": "Creator Table 5 open-response study.",
          "status": "verified",
          "value": 0.3
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-figqa-creator-open-response-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 10,
          "notes": "Creator Table 5 open-response study.",
          "status": "verified",
          "value": 0.3
        }
      ],
      "scope": {
        "filter": "Questions sampled and reworded to remove multiple-choice framing.",
        "n": 10,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "lab-bench-figqa-creator-open-response",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Open-response protocol is kept separate from multiple-choice results.",
        "status": "verified"
      },
      "work_id": "lab-bench-paper",
      "work_version_id": "lab-bench-paper-2024-07-14"
    },
    {
      "benchmark_id": "lab-bench-figqa",
      "benchmark_version": null,
      "comparability_group": "lab-bench-figqa-sonnet46-adaptive-max-crop-tool-five-runs",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-figqa-crop-tool-protocol",
          "locator": {
            "note": "Adaptive thinking, max effort, simple crop tool, five runs, 95% CI, and unreported n/snapshot/grader.",
            "type": "section",
            "value": "Section 2.17.1, printed page 33"
          },
          "source_id": "anthropic-sonnet-4-6-system-card",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-figqa-crop-tool-results",
          "locator": {
            "note": "All three crop-tool point estimates.",
            "type": "section",
            "value": "Section 2.17.1 and Figure 2.17.1.A, printed page 33"
          },
          "source_id": "anthropic-sonnet-4-6-system-card",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-figqa-crop-tool",
      "metrics": [
        {
          "aggregation": "mean over five runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "FigQA score",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "claude-opus-4-6",
        "claude-sonnet-4-5",
        "claude-sonnet-4-6"
      ],
      "protocol": {
        "contamination": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": null,
          "model": null,
          "reporting_status": "not_reported",
          "type": null
        },
        "reasoning": {
          "notes": "Shared setting for every registered model.",
          "reporting_status": "reported",
          "value": "adaptive thinking at max effort"
        },
        "repeats": {
          "notes": "Five evaluation runs.",
          "reporting_status": "reported",
          "value": 5
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "statistical": {
          "notes": "Numeric confidence-interval bounds are not printed.",
          "reporting_status": "reported",
          "value": "mean over five runs with 95% CI shown"
        },
        "system_prompt_public": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "No browser reported.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "No code execution reported.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No container reported.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "No database tools reported.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Crop-tool condition.",
            "reporting_status": "reported",
            "value": [
              "simple image cropping tool"
            ]
          },
          "internet": {
            "notes": "No internet reported.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "Multiple-choice figure questions.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-figqa-crop-tool-results"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-sonnet-4-6",
          "n": null,
          "notes": "95% CI is plotted but numeric bounds are not reported.",
          "status": "verified",
          "value": 77.1
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-figqa-crop-tool-results"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-sonnet-4-5",
          "n": null,
          "notes": "95% CI is plotted but numeric bounds are not reported.",
          "status": "verified",
          "value": 59.3
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-figqa-crop-tool-results"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-opus-4-6",
          "n": null,
          "notes": "95% CI is plotted but numeric bounds are not reported.",
          "status": "verified",
          "value": 78.3
        }
      ],
      "scope": {
        "filter": "FigQA track; benchmark snapshot, public/private scope, and realized item count are not reported.",
        "n": null,
        "reporting_status": "not_reported",
        "selection": null,
        "subset_id": null,
        "type": "track"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Kept separate because the image-cropping tool materially changes the protocol.",
        "status": "verified"
      },
      "work_id": "anthropic-sonnet-4-6-system-card",
      "work_version_id": "anthropic-sonnet-4-6-system-card-2026-02-17"
    },
    {
      "benchmark_id": "lab-bench-figqa",
      "benchmark_version": null,
      "comparability_group": "lab-bench-figqa-sonnet46-adaptive-max-no-tools-five-runs",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-figqa-no-tools-protocol",
          "locator": {
            "note": "Adaptive thinking, max effort, no tools, five runs, 95% CI, and unreported n/snapshot/grader.",
            "type": "section",
            "value": "Section 2.17.1, printed page 33"
          },
          "source_id": "anthropic-sonnet-4-6-system-card",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-figqa-no-tools-results",
          "locator": {
            "note": "All three no-tool point estimates.",
            "type": "section",
            "value": "Section 2.17.1 and Figure 2.17.1.A, printed page 33"
          },
          "source_id": "anthropic-sonnet-4-6-system-card",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-figqa-no-tools",
      "metrics": [
        {
          "aggregation": "mean over five runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "FigQA score",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "claude-opus-4-6",
        "claude-sonnet-4-5",
        "claude-sonnet-4-6"
      ],
      "protocol": {
        "contamination": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": null,
          "model": null,
          "reporting_status": "not_reported",
          "type": null
        },
        "reasoning": {
          "notes": "Shared setting for every registered model.",
          "reporting_status": "reported",
          "value": "adaptive thinking at max effort"
        },
        "repeats": {
          "notes": "Five evaluation runs.",
          "reporting_status": "reported",
          "value": 5
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "statistical": {
          "notes": "Numeric confidence-interval bounds are not printed.",
          "reporting_status": "reported",
          "value": "mean over five runs with 95% CI shown"
        },
        "system_prompt_public": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "No-tool condition.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "No-tool condition.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No-tool condition.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "No-tool condition.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "No-tool condition.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "No-tool condition.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "Multiple-choice figure questions.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-figqa-no-tools-results"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-sonnet-4-6",
          "n": null,
          "notes": "95% CI is plotted but numeric bounds are not reported.",
          "status": "verified",
          "value": 58.8
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-figqa-no-tools-results"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-sonnet-4-5",
          "n": null,
          "notes": "95% CI is plotted but numeric bounds are not reported.",
          "status": "verified",
          "value": 53.4
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-figqa-no-tools-results"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-opus-4-6",
          "n": null,
          "notes": "95% CI is plotted but numeric bounds are not reported.",
          "status": "verified",
          "value": 58.0
        }
      ],
      "scope": {
        "filter": "FigQA track; benchmark snapshot, public/private scope, and realized item count are not reported.",
        "n": null,
        "reporting_status": "not_reported",
        "selection": null,
        "subset_id": null,
        "type": "track"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Kept separate from the crop-tool condition; all point estimates and protocol claims were checked on printed page 33.",
        "status": "verified"
      },
      "work_id": "anthropic-sonnet-4-6-system-card",
      "work_version_id": "anthropic-sonnet-4-6-system-card-2026-02-17"
    },
    {
      "benchmark_id": "lab-bench-litqa2",
      "benchmark_version": "paper-v3",
      "comparability_group": "lab-bench-litqa2-paper-v3-zero-shot-cot-no-tools",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-litqa2-creator-mcq-protocol",
          "locator": {
            "note": "Full scope, zero-shot chain-of-thought prompt, no tools, three runs, exact metrics, parser, and model identities.",
            "type": "section",
            "value": "Appendix D.1–D.4; Table 1 and Appendix Table 6"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-litqa2-creator-mcq-results",
          "locator": {
            "note": "Accuracy, precision, and coverage. Human row excluded.",
            "type": "table",
            "value": "Tables 2–4, row LitQA2"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-litqa2-creator-mcq",
      "metrics": [
        {
          "aggregation": "correct / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Accuracy",
          "tolerance": "exact choice",
          "unit": "proportion"
        },
        {
          "aggregation": "correct / attempted questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Precision (selective accuracy)",
          "tolerance": "insufficient-information responses excluded from denominator",
          "unit": "proportion"
        },
        {
          "aggregation": "attempted / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "coverage",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Coverage",
          "tolerance": "insufficient-information responses treated as not attempted",
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "lab-bench-claude-3-5-sonnet-20240620",
        "lab-bench-claude-3-haiku-20240307",
        "lab-bench-claude-3-opus-20240229",
        "lab-bench-gemini-1-5-pro-001",
        "lab-bench-gpt-4-turbo-unversioned",
        "lab-bench-gpt-4o-unversioned",
        "lab-bench-meta-llama-3-70b-instruct"
      ],
      "protocol": {
        "contamination": {
          "notes": "Main results combine both splits.",
          "reporting_status": "reported",
          "value": "full pre-release snapshot with 80% later public and 20% private contamination-monitoring splits"
        },
        "grader": {
          "human_review": false,
          "model": "Claude 2 fallback parser",
          "reporting_status": "reported",
          "type": "regex multiple-choice parser with Claude 2 fallback"
        },
        "reasoning": {
          "notes": "Prompt includes Think step by step.",
          "reporting_status": "reported",
          "value": "zero-shot chain-of-thought"
        },
        "repeats": {
          "notes": "Tables 2–4 average three runs per model.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public zero-shot prompt.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "No numeric confidence intervals are reported in Tables 2–4.",
          "reporting_status": "reported",
          "value": "mean across three runs per model"
        },
        "system_prompt_public": {
          "notes": "Appendix D.1 publishes the complete prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No common token budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One multiple-choice completion per question.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-litqa2-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 248,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.06
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-litqa2-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 248,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.47
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-litqa2-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 248,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.12
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-litqa2-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 248,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.06
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-litqa2-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 248,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.43
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-litqa2-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 248,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.14
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-litqa2-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 248,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.1
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-litqa2-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 248,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.44
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-litqa2-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 248,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.23
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-litqa2-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 248,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.27
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-litqa2-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 248,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.46
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-litqa2-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 248,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.58
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-litqa2-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 248,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.12
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-litqa2-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 248,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.4
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-litqa2-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 248,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.31
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-litqa2-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 248,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.27
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-litqa2-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 248,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.38
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-litqa2-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 248,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.7
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-litqa2-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 248,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.35
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-litqa2-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 248,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.38
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-litqa2-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 248,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.92
        }
      ],
      "scope": {
        "filter": "Complete creator snapshot, combining public and private splits.",
        "n": 248,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "All registered values were parsed from creator Tables 2–4 and independently checked against the HTML table rows.",
        "status": "verified"
      },
      "work_id": "lab-bench-paper",
      "work_version_id": "lab-bench-paper-2024-07-14"
    },
    {
      "benchmark_id": "lab-bench-protocolqa",
      "benchmark_version": null,
      "comparability_group": "lab-bench-protocolqa-anthropic-life-sciences-10shot-no-tools",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-protocolqa-anthropic-protocol",
          "locator": {
            "note": "10-shot, no search/bioinformatics tools, point estimates with error bars, and unreported n/repeats/grader.",
            "type": "section",
            "value": "Section 9.2.4.4 and Figure 9.2.4.4.A, printed pages 132–133"
          },
          "source_id": "anthropic-sonnet-4-5-system-card",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-protocolqa-anthropic-results",
          "locator": {
            "note": "Sonnet 4.5 score 0.83, Sonnet 4 score 0.74, and 10-shot prompting.",
            "type": "section",
            "value": "Making Claude a better research partner; footnote 1"
          },
          "source_id": "anthropic-life-sciences",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-protocolqa-anthropic",
      "metrics": [
        {
          "aggregation": null,
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "ProtocolQA score",
          "tolerance": null,
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "claude-sonnet-4",
        "claude-sonnet-4-5"
      ],
      "protocol": {
        "contamination": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": null,
          "model": null,
          "reporting_status": "not_reported",
          "type": null
        },
        "reasoning": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Official page footnote and linked system card report 10-shot prompting.",
          "reporting_status": "reported",
          "value": 10
        },
        "statistical": {
          "notes": "Repeat count and numeric interval bounds are not reported.",
          "reporting_status": "reported",
          "value": "reported point estimate with plotted error bars"
        },
        "system_prompt_public": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Linked system card says search tools were not included.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Linked system card says tools were not included in the testing environment.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Linked system card says tools were not included in the testing environment.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "Linked system card says bioinformatics tools were not included.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "No search or bioinformatics tools.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Linked system card says search tools were not included.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "Multiple-choice evaluation.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-protocolqa-anthropic-results"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-sonnet-4-5",
          "n": null,
          "notes": "Official release rounds the system-card point label 0.833 to 0.83.",
          "status": "verified",
          "value": 0.83
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-protocolqa-anthropic-results"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-sonnet-4",
          "n": null,
          "notes": "Official release rounds the system-card point label 0.741 to 0.74.",
          "status": "verified",
          "value": 0.74
        }
      ],
      "scope": {
        "filter": "ProtocolQA track; benchmark snapshot, public/private scope, and realized item count are not reported.",
        "n": null,
        "reporting_status": "not_reported",
        "selection": null,
        "subset_id": null,
        "type": "track"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "The public release gives rounded Sonnet results; the linked system-card figure gives three-decimal labels and the shared protocol.",
        "status": "verified"
      },
      "work_id": "anthropic-life-sciences",
      "work_version_id": "anthropic-life-sciences-2025-10-20"
    },
    {
      "benchmark_id": "lab-bench-protocolqa",
      "benchmark_version": null,
      "comparability_group": "lab-bench-protocolqa-anthropic-10shot-no-tools",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-protocolqa-anthropic-sonnet45-system-card-protocol",
          "locator": {
            "note": "Four selected tracks, 10-shot prompting, no search/bioinformatics tools, threshold, and unreported scope/repeats/grader.",
            "type": "section",
            "value": "Section 9.2.4.4, printed pages 132–133"
          },
          "source_id": "anthropic-sonnet-4-5-system-card",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-protocolqa-anthropic-sonnet45-system-card-results",
          "locator": {
            "note": "Exact point labels. Protocol Sonnet 4/4.5 labels are represented in the Claude for Life Sciences run to avoid duplicate result rows.",
            "type": "figure",
            "value": "Figure 9.2.4.4.A, printed page 133"
          },
          "source_id": "anthropic-sonnet-4-5-system-card",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-protocolqa-anthropic-sonnet45-system-card",
      "metrics": [
        {
          "aggregation": null,
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "LAB-Bench score",
          "tolerance": null,
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "claude-opus-4",
        "claude-opus-4-1"
      ],
      "protocol": {
        "contamination": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": null,
          "model": null,
          "reporting_status": "not_reported",
          "type": null
        },
        "reasoning": {
          "notes": "Thinking/effort configuration is not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "repeats": {
          "notes": "The number of repeats underlying the plotted error bars is not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Figure title and caption specify 10-shot prompting.",
          "reporting_status": "reported",
          "value": 10
        },
        "statistical": {
          "notes": "The figure does not give numeric interval bounds.",
          "reporting_status": "reported",
          "value": "reported point estimate with plotted error bars"
        },
        "system_prompt_public": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "The system card says search and bioinformatics tools were not included in the testing environment.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "The system card says search and bioinformatics tools were not included in the testing environment.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "The system card says search and bioinformatics tools were not included in the testing environment.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "The system card says search and bioinformatics tools were not included in the testing environment.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "The system card says search and bioinformatics tools were not included in the testing environment.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "The system card says search and bioinformatics tools were not included in the testing environment.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "Multiple-choice evaluation.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-protocolqa-anthropic-sonnet45-system-card-results"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-opus-4",
          "n": null,
          "notes": "Point label transcribed from Figure 9.2.4.4.A; numeric error-bar bounds are not reported.",
          "status": "verified",
          "value": 0.796
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-protocolqa-anthropic-sonnet45-system-card-results"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-opus-4-1",
          "n": null,
          "notes": "Point label transcribed from Figure 9.2.4.4.A; numeric error-bar bounds are not reported.",
          "status": "verified",
          "value": 0.833
        }
      ],
      "scope": {
        "filter": "lab-bench-protocolqa track; benchmark snapshot, public/private scope, and realized n are not reported.",
        "n": null,
        "reporting_status": "not_reported",
        "selection": null,
        "subset_id": null,
        "type": "track"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Protocol, exact point labels, and missing n/repeat details were checked on printed pages 132–133.",
        "status": "verified"
      },
      "work_id": "anthropic-sonnet-4-5-system-card",
      "work_version_id": "anthropic-sonnet-4-5-system-card-2025-09-29"
    },
    {
      "benchmark_id": "lab-bench-protocolqa",
      "benchmark_version": "paper-v3",
      "comparability_group": "lab-bench-protocolqa-paper-v3-zero-shot-cot-no-tools",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-protocolqa-creator-mcq-protocol",
          "locator": {
            "note": "Full scope, zero-shot chain-of-thought prompt, no tools, three runs, exact metrics, parser, and model identities.",
            "type": "section",
            "value": "Appendix D.1–D.4; Table 1 and Appendix Table 6"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-protocolqa-creator-mcq-results",
          "locator": {
            "note": "Accuracy, precision, and coverage. Human row excluded.",
            "type": "table",
            "value": "Tables 2–4, row ProtocolQA"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-protocolqa-creator-mcq",
      "metrics": [
        {
          "aggregation": "correct / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Accuracy",
          "tolerance": "exact choice",
          "unit": "proportion"
        },
        {
          "aggregation": "correct / attempted questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Precision (selective accuracy)",
          "tolerance": "insufficient-information responses excluded from denominator",
          "unit": "proportion"
        },
        {
          "aggregation": "attempted / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "coverage",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Coverage",
          "tolerance": "insufficient-information responses treated as not attempted",
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "lab-bench-claude-3-5-sonnet-20240620",
        "lab-bench-claude-3-haiku-20240307",
        "lab-bench-claude-3-opus-20240229",
        "lab-bench-gemini-1-5-pro-001",
        "lab-bench-gpt-4-turbo-unversioned",
        "lab-bench-gpt-4o-unversioned",
        "lab-bench-meta-llama-3-70b-instruct"
      ],
      "protocol": {
        "contamination": {
          "notes": "Main results combine both splits.",
          "reporting_status": "reported",
          "value": "full pre-release snapshot with 80% later public and 20% private contamination-monitoring splits"
        },
        "grader": {
          "human_review": false,
          "model": "Claude 2 fallback parser",
          "reporting_status": "reported",
          "type": "regex multiple-choice parser with Claude 2 fallback"
        },
        "reasoning": {
          "notes": "Prompt includes Think step by step.",
          "reporting_status": "reported",
          "value": "zero-shot chain-of-thought"
        },
        "repeats": {
          "notes": "Tables 2–4 average three runs per model.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public zero-shot prompt.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "No numeric confidence intervals are reported in Tables 2–4.",
          "reporting_status": "reported",
          "value": "mean across three runs per model"
        },
        "system_prompt_public": {
          "notes": "Appendix D.1 publishes the complete prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No common token budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One multiple-choice completion per question.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-protocolqa-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 135,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.48
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-protocolqa-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 135,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.66
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-protocolqa-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 135,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.73
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-protocolqa-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 135,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.52
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-protocolqa-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 135,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.62
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-protocolqa-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 135,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.84
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-protocolqa-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 135,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.47
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-protocolqa-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 135,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.58
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-protocolqa-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 135,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.81
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-protocolqa-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 135,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.53
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-protocolqa-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 135,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.56
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-protocolqa-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 135,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.95
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-protocolqa-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 135,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.37
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-protocolqa-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 135,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.59
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-protocolqa-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 135,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.62
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-protocolqa-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 135,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.45
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-protocolqa-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 135,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.51
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-protocolqa-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 135,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.87
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-protocolqa-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 135,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.44
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-protocolqa-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 135,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.49
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-protocolqa-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 135,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.9
        }
      ],
      "scope": {
        "filter": "Complete creator snapshot, combining public and private splits.",
        "n": 135,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "All registered values were parsed from creator Tables 2–4 and independently checked against the HTML table rows.",
        "status": "verified"
      },
      "work_id": "lab-bench-paper",
      "work_version_id": "lab-bench-paper-2024-07-14"
    },
    {
      "benchmark_id": "lab-bench-protocolqa",
      "benchmark_version": "paper-v3",
      "comparability_group": "lab-bench-protocolqa-paper-v3-open-response",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-protocolqa-creator-open-response-protocol",
          "locator": {
            "note": "Subset construction, modified wording, expert grading, and second review.",
            "type": "section",
            "value": "Section 2.4 and Appendix B.2"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-protocolqa-creator-open-response-results",
          "locator": {
            "note": "lab-bench-protocolqa model accuracy and question count.",
            "type": "table",
            "value": "Table 5"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-protocolqa-creator-open-response",
      "metrics": [
        {
          "aggregation": "correct / sampled modified questions",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Accuracy",
          "tolerance": "expert judgment against ideal multiple-choice answer",
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "lab-bench-claude-3-5-sonnet-20240620",
        "lab-bench-gpt-4o-unversioned"
      ],
      "protocol": {
        "contamination": {
          "notes": "Main results combine both splits.",
          "reporting_status": "reported",
          "value": "full pre-release snapshot with 80% later public and 20% private contamination-monitoring splits"
        },
        "grader": {
          "human_review": true,
          "model": null,
          "reporting_status": "reported",
          "type": "expert-biologist manual grading against the ideal answer, reviewed by a second expert"
        },
        "reasoning": {
          "notes": "Prompt includes Think step by step.",
          "reporting_status": "reported",
          "value": "zero-shot chain-of-thought"
        },
        "repeats": {
          "notes": "Repeat count is not reported for Table 5.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The paper calls this a small-scale open-answer run but does not restate shot count for the modified prompt.",
          "reporting_status": "not_reported",
          "value": null
        },
        "statistical": {
          "notes": "No confidence intervals reported.",
          "reporting_status": "reported",
          "value": "single reported accuracy per model"
        },
        "system_prompt_public": {
          "notes": "Appendix D.1 publishes the complete prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No common token budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One multiple-choice completion per question.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-protocolqa-creator-open-response-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 20,
          "notes": "Creator Table 5 open-response study.",
          "status": "verified",
          "value": 0.3
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-protocolqa-creator-open-response-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 20,
          "notes": "Creator Table 5 open-response study.",
          "status": "verified",
          "value": 0.2
        }
      ],
      "scope": {
        "filter": "Questions sampled and reworded to remove multiple-choice framing.",
        "n": 20,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": "lab-bench-protocolqa-creator-open-response",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Open-response protocol is kept separate from multiple-choice results.",
        "status": "verified"
      },
      "work_id": "lab-bench-paper",
      "work_version_id": "lab-bench-paper-2024-07-14"
    },
    {
      "benchmark_id": "lab-bench-seqqa",
      "benchmark_version": null,
      "comparability_group": "lab-bench-seqqa-anthropic-10shot-no-tools",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-anthropic-sonnet45-system-card-protocol",
          "locator": {
            "note": "Four selected tracks, 10-shot prompting, no search/bioinformatics tools, threshold, and unreported scope/repeats/grader.",
            "type": "section",
            "value": "Section 9.2.4.4, printed pages 132–133"
          },
          "source_id": "anthropic-sonnet-4-5-system-card",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-anthropic-sonnet45-system-card-results",
          "locator": {
            "note": "Exact point labels. Protocol Sonnet 4/4.5 labels are represented in the Claude for Life Sciences run to avoid duplicate result rows.",
            "type": "figure",
            "value": "Figure 9.2.4.4.A, printed page 133"
          },
          "source_id": "anthropic-sonnet-4-5-system-card",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-seqqa-anthropic-sonnet45-system-card",
      "metrics": [
        {
          "aggregation": null,
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "LAB-Bench score",
          "tolerance": null,
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "claude-opus-4",
        "claude-opus-4-1",
        "claude-sonnet-4",
        "claude-sonnet-4-5"
      ],
      "protocol": {
        "contamination": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": null,
          "model": null,
          "reporting_status": "not_reported",
          "type": null
        },
        "reasoning": {
          "notes": "Thinking/effort configuration is not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "repeats": {
          "notes": "The number of repeats underlying the plotted error bars is not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Figure title and caption specify 10-shot prompting.",
          "reporting_status": "reported",
          "value": 10
        },
        "statistical": {
          "notes": "The figure does not give numeric interval bounds.",
          "reporting_status": "reported",
          "value": "reported point estimate with plotted error bars"
        },
        "system_prompt_public": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "The system card says search and bioinformatics tools were not included in the testing environment.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "The system card says search and bioinformatics tools were not included in the testing environment.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "The system card says search and bioinformatics tools were not included in the testing environment.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "The system card says search and bioinformatics tools were not included in the testing environment.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "The system card says search and bioinformatics tools were not included in the testing environment.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "The system card says search and bioinformatics tools were not included in the testing environment.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "Multiple-choice evaluation.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-anthropic-sonnet45-system-card-results"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-sonnet-4",
          "n": null,
          "notes": "Point label transcribed from Figure 9.2.4.4.A; numeric error-bar bounds are not reported.",
          "status": "verified",
          "value": 0.682
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-anthropic-sonnet45-system-card-results"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-opus-4",
          "n": null,
          "notes": "Point label transcribed from Figure 9.2.4.4.A; numeric error-bar bounds are not reported.",
          "status": "verified",
          "value": 0.723
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-anthropic-sonnet45-system-card-results"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-opus-4-1",
          "n": null,
          "notes": "Point label transcribed from Figure 9.2.4.4.A; numeric error-bar bounds are not reported.",
          "status": "verified",
          "value": 0.785
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-anthropic-sonnet45-system-card-results"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-sonnet-4-5",
          "n": null,
          "notes": "Point label transcribed from Figure 9.2.4.4.A; numeric error-bar bounds are not reported.",
          "status": "verified",
          "value": 0.78
        }
      ],
      "scope": {
        "filter": "lab-bench-seqqa track; benchmark snapshot, public/private scope, and realized n are not reported.",
        "n": null,
        "reporting_status": "not_reported",
        "selection": null,
        "subset_id": null,
        "type": "track"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Protocol, exact point labels, and missing n/repeat details were checked on printed pages 132–133.",
        "status": "verified"
      },
      "work_id": "anthropic-sonnet-4-5-system-card",
      "work_version_id": "anthropic-sonnet-4-5-system-card-2025-09-29"
    },
    {
      "benchmark_id": "lab-bench-seqqa-orf-seq-aaid",
      "benchmark_version": "paper-v3",
      "comparability_group": "lab-bench-seqqa-orf-seq-aaid-paper-v3-zero-shot-cot-no-tools",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-orf-seq-aaid-creator-mcq-protocol",
          "locator": {
            "note": "Full scope, zero-shot chain-of-thought prompt, no tools, three runs, exact metrics, parser, and model identities.",
            "type": "section",
            "value": "Appendix D.1–D.4; Table 1 and Appendix Table 6"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-orf-seq-aaid-creator-mcq-results",
          "locator": {
            "note": "Accuracy, precision, and coverage. Human row excluded.",
            "type": "table",
            "value": "Tables 2–4, row SeqQA_ORF-seq-AAid-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-seqqa-orf-seq-aaid-creator-mcq",
      "metrics": [
        {
          "aggregation": "correct / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Accuracy",
          "tolerance": "exact choice",
          "unit": "proportion"
        },
        {
          "aggregation": "correct / attempted questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Precision (selective accuracy)",
          "tolerance": "insufficient-information responses excluded from denominator",
          "unit": "proportion"
        },
        {
          "aggregation": "attempted / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "coverage",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Coverage",
          "tolerance": "insufficient-information responses treated as not attempted",
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "lab-bench-claude-3-5-sonnet-20240620",
        "lab-bench-claude-3-haiku-20240307",
        "lab-bench-claude-3-opus-20240229",
        "lab-bench-gemini-1-5-pro-001",
        "lab-bench-gpt-4-turbo-unversioned",
        "lab-bench-gpt-4o-unversioned",
        "lab-bench-meta-llama-3-70b-instruct"
      ],
      "protocol": {
        "contamination": {
          "notes": "Main results combine both splits.",
          "reporting_status": "reported",
          "value": "full pre-release snapshot with 80% later public and 20% private contamination-monitoring splits"
        },
        "grader": {
          "human_review": false,
          "model": "Claude 2 fallback parser",
          "reporting_status": "reported",
          "type": "regex multiple-choice parser with Claude 2 fallback"
        },
        "reasoning": {
          "notes": "Prompt includes Think step by step.",
          "reporting_status": "reported",
          "value": "zero-shot chain-of-thought"
        },
        "repeats": {
          "notes": "Tables 2–4 average three runs per model.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public zero-shot prompt.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "No numeric confidence intervals are reported in Tables 2–4.",
          "reporting_status": "reported",
          "value": "mean across three runs per model"
        },
        "system_prompt_public": {
          "notes": "Appendix D.1 publishes the complete prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No common token budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One multiple-choice completion per question.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaid-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.07
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaid-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.13
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaid-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.57
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaid-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.03
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaid-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.13
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaid-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.2
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaid-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.02
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaid-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.09
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaid-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.23
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaid-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.16
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaid-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.23
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaid-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.69
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaid-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.07
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaid-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.15
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaid-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.47
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaid-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.17
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaid-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.19
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaid-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.91
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaid-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.19
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaid-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.19
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaid-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.97
        }
      ],
      "scope": {
        "filter": "Complete creator snapshot, combining public and private splits.",
        "n": 50,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "All registered values were parsed from creator Tables 2–4 and independently checked against the HTML table rows.",
        "status": "verified"
      },
      "work_id": "lab-bench-paper",
      "work_version_id": "lab-bench-paper-2024-07-14"
    },
    {
      "benchmark_id": "lab-bench-seqqa-orf-seq-aaseq",
      "benchmark_version": "paper-v3",
      "comparability_group": "lab-bench-seqqa-orf-seq-aaseq-paper-v3-zero-shot-cot-no-tools",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-orf-seq-aaseq-creator-mcq-protocol",
          "locator": {
            "note": "Full scope, zero-shot chain-of-thought prompt, no tools, three runs, exact metrics, parser, and model identities.",
            "type": "section",
            "value": "Appendix D.1–D.4; Table 1 and Appendix Table 6"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-orf-seq-aaseq-creator-mcq-results",
          "locator": {
            "note": "Accuracy, precision, and coverage. Human row excluded.",
            "type": "table",
            "value": "Tables 2–4, row SeqQA_ORF-seq-AAseq-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-seqqa-orf-seq-aaseq-creator-mcq",
      "metrics": [
        {
          "aggregation": "correct / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Accuracy",
          "tolerance": "exact choice",
          "unit": "proportion"
        },
        {
          "aggregation": "correct / attempted questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Precision (selective accuracy)",
          "tolerance": "insufficient-information responses excluded from denominator",
          "unit": "proportion"
        },
        {
          "aggregation": "attempted / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "coverage",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Coverage",
          "tolerance": "insufficient-information responses treated as not attempted",
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "lab-bench-claude-3-5-sonnet-20240620",
        "lab-bench-claude-3-haiku-20240307",
        "lab-bench-claude-3-opus-20240229",
        "lab-bench-gemini-1-5-pro-001",
        "lab-bench-gpt-4-turbo-unversioned",
        "lab-bench-gpt-4o-unversioned",
        "lab-bench-meta-llama-3-70b-instruct"
      ],
      "protocol": {
        "contamination": {
          "notes": "Main results combine both splits.",
          "reporting_status": "reported",
          "value": "full pre-release snapshot with 80% later public and 20% private contamination-monitoring splits"
        },
        "grader": {
          "human_review": false,
          "model": "Claude 2 fallback parser",
          "reporting_status": "reported",
          "type": "regex multiple-choice parser with Claude 2 fallback"
        },
        "reasoning": {
          "notes": "Prompt includes Think step by step.",
          "reporting_status": "reported",
          "value": "zero-shot chain-of-thought"
        },
        "repeats": {
          "notes": "Tables 2–4 average three runs per model.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public zero-shot prompt.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "No numeric confidence intervals are reported in Tables 2–4.",
          "reporting_status": "reported",
          "value": "mean across three runs per model"
        },
        "system_prompt_public": {
          "notes": "Appendix D.1 publishes the complete prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No common token budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One multiple-choice completion per question.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaseq-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.87
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaseq-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.89
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaseq-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.99
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaseq-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.67
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaseq-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.7
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaseq-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.95
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaseq-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.41
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaseq-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.67
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaseq-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.61
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaseq-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.5
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaseq-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.5
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaseq-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.99
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaseq-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.39
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaseq-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.5
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaseq-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.77
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaseq-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.57
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaseq-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.57
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaseq-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 1.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaseq-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.49
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaseq-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.49
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-aaseq-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 1.0
        }
      ],
      "scope": {
        "filter": "Complete creator snapshot, combining public and private splits.",
        "n": 50,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "All registered values were parsed from creator Tables 2–4 and independently checked against the HTML table rows.",
        "status": "verified"
      },
      "work_id": "lab-bench-paper",
      "work_version_id": "lab-bench-paper-2024-07-14"
    },
    {
      "benchmark_id": "lab-bench-seqqa-orf-seq-numlen",
      "benchmark_version": "paper-v3",
      "comparability_group": "lab-bench-seqqa-orf-seq-numlen-paper-v3-zero-shot-cot-no-tools",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-orf-seq-numlen-creator-mcq-protocol",
          "locator": {
            "note": "Full scope, zero-shot chain-of-thought prompt, no tools, three runs, exact metrics, parser, and model identities.",
            "type": "section",
            "value": "Appendix D.1–D.4; Table 1 and Appendix Table 6"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-orf-seq-numlen-creator-mcq-results",
          "locator": {
            "note": "Accuracy, precision, and coverage. Human row excluded.",
            "type": "table",
            "value": "Tables 2–4, row SeqQA_ORF-seq-numlen-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-seqqa-orf-seq-numlen-creator-mcq",
      "metrics": [
        {
          "aggregation": "correct / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Accuracy",
          "tolerance": "exact choice",
          "unit": "proportion"
        },
        {
          "aggregation": "correct / attempted questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Precision (selective accuracy)",
          "tolerance": "insufficient-information responses excluded from denominator",
          "unit": "proportion"
        },
        {
          "aggregation": "attempted / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "coverage",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Coverage",
          "tolerance": "insufficient-information responses treated as not attempted",
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "lab-bench-claude-3-5-sonnet-20240620",
        "lab-bench-claude-3-haiku-20240307",
        "lab-bench-claude-3-opus-20240229",
        "lab-bench-gemini-1-5-pro-001",
        "lab-bench-gpt-4-turbo-unversioned",
        "lab-bench-gpt-4o-unversioned",
        "lab-bench-meta-llama-3-70b-instruct"
      ],
      "protocol": {
        "contamination": {
          "notes": "Main results combine both splits.",
          "reporting_status": "reported",
          "value": "full pre-release snapshot with 80% later public and 20% private contamination-monitoring splits"
        },
        "grader": {
          "human_review": false,
          "model": "Claude 2 fallback parser",
          "reporting_status": "reported",
          "type": "regex multiple-choice parser with Claude 2 fallback"
        },
        "reasoning": {
          "notes": "Prompt includes Think step by step.",
          "reporting_status": "reported",
          "value": "zero-shot chain-of-thought"
        },
        "repeats": {
          "notes": "Tables 2–4 average three runs per model.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public zero-shot prompt.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "No numeric confidence intervals are reported in Tables 2–4.",
          "reporting_status": "reported",
          "value": "mean across three runs per model"
        },
        "system_prompt_public": {
          "notes": "Appendix D.1 publishes the complete prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No common token budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One multiple-choice completion per question.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-numlen-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-numlen-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-numlen-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.01
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-numlen-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.07
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-numlen-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.28
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-numlen-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.26
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-numlen-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.05
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-numlen-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.18
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-numlen-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.31
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-numlen-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.11
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-numlen-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.14
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-numlen-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.81
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-numlen-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.07
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-numlen-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.27
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-numlen-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.29
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-numlen-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.26
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-numlen-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.26
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-numlen-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 1.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-numlen-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.23
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-numlen-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.29
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-seq-numlen-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.81
        }
      ],
      "scope": {
        "filter": "Complete creator snapshot, combining public and private splits.",
        "n": 50,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "All registered values were parsed from creator Tables 2–4 and independently checked against the HTML table rows.",
        "status": "verified"
      },
      "work_id": "lab-bench-paper",
      "work_version_id": "lab-bench-paper-2024-07-14"
    },
    {
      "benchmark_id": "lab-bench-seqqa-orf-transeff",
      "benchmark_version": "paper-v3",
      "comparability_group": "lab-bench-seqqa-orf-transeff-paper-v3-zero-shot-cot-no-tools",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-orf-transeff-creator-mcq-protocol",
          "locator": {
            "note": "Full scope, zero-shot chain-of-thought prompt, no tools, three runs, exact metrics, parser, and model identities.",
            "type": "section",
            "value": "Appendix D.1–D.4; Table 1 and Appendix Table 6"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-orf-transeff-creator-mcq-results",
          "locator": {
            "note": "Accuracy, precision, and coverage. Human row excluded.",
            "type": "table",
            "value": "Tables 2–4, row SeqQA_ORF-transeff-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-seqqa-orf-transeff-creator-mcq",
      "metrics": [
        {
          "aggregation": "correct / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Accuracy",
          "tolerance": "exact choice",
          "unit": "proportion"
        },
        {
          "aggregation": "correct / attempted questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Precision (selective accuracy)",
          "tolerance": "insufficient-information responses excluded from denominator",
          "unit": "proportion"
        },
        {
          "aggregation": "attempted / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "coverage",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Coverage",
          "tolerance": "insufficient-information responses treated as not attempted",
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "lab-bench-claude-3-5-sonnet-20240620",
        "lab-bench-claude-3-haiku-20240307",
        "lab-bench-claude-3-opus-20240229",
        "lab-bench-gemini-1-5-pro-001",
        "lab-bench-gpt-4-turbo-unversioned",
        "lab-bench-gpt-4o-unversioned",
        "lab-bench-meta-llama-3-70b-instruct"
      ],
      "protocol": {
        "contamination": {
          "notes": "Main results combine both splits.",
          "reporting_status": "reported",
          "value": "full pre-release snapshot with 80% later public and 20% private contamination-monitoring splits"
        },
        "grader": {
          "human_review": false,
          "model": "Claude 2 fallback parser",
          "reporting_status": "reported",
          "type": "regex multiple-choice parser with Claude 2 fallback"
        },
        "reasoning": {
          "notes": "Prompt includes Think step by step.",
          "reporting_status": "reported",
          "value": "zero-shot chain-of-thought"
        },
        "repeats": {
          "notes": "Tables 2–4 average three runs per model.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public zero-shot prompt.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "No numeric confidence intervals are reported in Tables 2–4.",
          "reporting_status": "reported",
          "value": "mean across three runs per model"
        },
        "system_prompt_public": {
          "notes": "Appendix D.1 publishes the complete prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No common token budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One multiple-choice completion per question.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-transeff-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.87
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-transeff-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.92
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-transeff-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.95
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-transeff-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.64
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-transeff-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.71
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-transeff-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.9
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-transeff-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.47
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-transeff-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.67
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-transeff-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.71
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-transeff-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.48
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-transeff-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.66
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-transeff-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.73
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-transeff-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.29
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-transeff-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.49
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-transeff-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.59
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-transeff-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.47
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-transeff-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.52
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-transeff-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.92
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-transeff-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.45
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-transeff-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.46
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-orf-transeff-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.98
        }
      ],
      "scope": {
        "filter": "Complete creator snapshot, combining public and private splits.",
        "n": 50,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "All registered values were parsed from creator Tables 2–4 and independently checked against the HTML table rows.",
        "status": "verified"
      },
      "work_id": "lab-bench-paper",
      "work_version_id": "lab-bench-paper-2024-07-14"
    },
    {
      "benchmark_id": "lab-bench-seqqa-pcr-gene-enzprimers",
      "benchmark_version": "paper-v3",
      "comparability_group": "lab-bench-seqqa-pcr-gene-enzprimers-paper-v3-zero-shot-cot-no-tools",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-pcr-gene-enzprimers-creator-mcq-protocol",
          "locator": {
            "note": "Full scope, zero-shot chain-of-thought prompt, no tools, three runs, exact metrics, parser, and model identities.",
            "type": "section",
            "value": "Appendix D.1–D.4; Table 1 and Appendix Table 6"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-pcr-gene-enzprimers-creator-mcq-results",
          "locator": {
            "note": "Accuracy, precision, and coverage. Human row excluded.",
            "type": "table",
            "value": "Tables 2–4, row SeqQA_PCR-gene-enzprimers-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-seqqa-pcr-gene-enzprimers-creator-mcq",
      "metrics": [
        {
          "aggregation": "correct / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Accuracy",
          "tolerance": "exact choice",
          "unit": "proportion"
        },
        {
          "aggregation": "correct / attempted questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Precision (selective accuracy)",
          "tolerance": "insufficient-information responses excluded from denominator",
          "unit": "proportion"
        },
        {
          "aggregation": "attempted / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "coverage",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Coverage",
          "tolerance": "insufficient-information responses treated as not attempted",
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "lab-bench-claude-3-5-sonnet-20240620",
        "lab-bench-claude-3-haiku-20240307",
        "lab-bench-claude-3-opus-20240229",
        "lab-bench-gemini-1-5-pro-001",
        "lab-bench-gpt-4-turbo-unversioned",
        "lab-bench-gpt-4o-unversioned",
        "lab-bench-meta-llama-3-70b-instruct"
      ],
      "protocol": {
        "contamination": {
          "notes": "Main results combine both splits.",
          "reporting_status": "reported",
          "value": "full pre-release snapshot with 80% later public and 20% private contamination-monitoring splits"
        },
        "grader": {
          "human_review": false,
          "model": "Claude 2 fallback parser",
          "reporting_status": "reported",
          "type": "regex multiple-choice parser with Claude 2 fallback"
        },
        "reasoning": {
          "notes": "Prompt includes Think step by step.",
          "reporting_status": "reported",
          "value": "zero-shot chain-of-thought"
        },
        "repeats": {
          "notes": "Tables 2–4 average three runs per model.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public zero-shot prompt.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "No numeric confidence intervals are reported in Tables 2–4.",
          "reporting_status": "reported",
          "value": "mean across three runs per model"
        },
        "system_prompt_public": {
          "notes": "Appendix D.1 publishes the complete prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No common token budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One multiple-choice completion per question.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-enzprimers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.57
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-enzprimers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.78
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-enzprimers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.72
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-enzprimers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.75
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-enzprimers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.87
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-enzprimers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.86
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-enzprimers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.47
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-enzprimers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.8
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-enzprimers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.59
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-enzprimers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.6
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-enzprimers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.64
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-enzprimers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.93
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-enzprimers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.49
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-enzprimers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.6
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-enzprimers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.81
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-enzprimers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.61
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-enzprimers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.61
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-enzprimers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 1.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-enzprimers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.58
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-enzprimers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.58
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-enzprimers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.99
        }
      ],
      "scope": {
        "filter": "Complete creator snapshot, combining public and private splits.",
        "n": 50,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "All registered values were parsed from creator Tables 2–4 and independently checked against the HTML table rows.",
        "status": "verified"
      },
      "work_id": "lab-bench-paper",
      "work_version_id": "lab-bench-paper-2024-07-14"
    },
    {
      "benchmark_id": "lab-bench-seqqa-pcr-gene-gibshindprimers",
      "benchmark_version": "paper-v3",
      "comparability_group": "lab-bench-seqqa-pcr-gene-gibshindprimers-paper-v3-zero-shot-cot-no-tools",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-pcr-gene-gibshindprimers-creator-mcq-protocol",
          "locator": {
            "note": "Full scope, zero-shot chain-of-thought prompt, no tools, three runs, exact metrics, parser, and model identities.",
            "type": "section",
            "value": "Appendix D.1–D.4; Table 1 and Appendix Table 6"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-pcr-gene-gibshindprimers-creator-mcq-results",
          "locator": {
            "note": "Accuracy, precision, and coverage. Human row excluded.",
            "type": "table",
            "value": "Tables 2–4, row SeqQA_PCR-gene-gibshindprimers-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-seqqa-pcr-gene-gibshindprimers-creator-mcq",
      "metrics": [
        {
          "aggregation": "correct / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Accuracy",
          "tolerance": "exact choice",
          "unit": "proportion"
        },
        {
          "aggregation": "correct / attempted questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Precision (selective accuracy)",
          "tolerance": "insufficient-information responses excluded from denominator",
          "unit": "proportion"
        },
        {
          "aggregation": "attempted / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "coverage",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Coverage",
          "tolerance": "insufficient-information responses treated as not attempted",
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "lab-bench-claude-3-5-sonnet-20240620",
        "lab-bench-claude-3-haiku-20240307",
        "lab-bench-claude-3-opus-20240229",
        "lab-bench-gemini-1-5-pro-001",
        "lab-bench-gpt-4-turbo-unversioned",
        "lab-bench-gpt-4o-unversioned",
        "lab-bench-meta-llama-3-70b-instruct"
      ],
      "protocol": {
        "contamination": {
          "notes": "Main results combine both splits.",
          "reporting_status": "reported",
          "value": "full pre-release snapshot with 80% later public and 20% private contamination-monitoring splits"
        },
        "grader": {
          "human_review": false,
          "model": "Claude 2 fallback parser",
          "reporting_status": "reported",
          "type": "regex multiple-choice parser with Claude 2 fallback"
        },
        "reasoning": {
          "notes": "Prompt includes Think step by step.",
          "reporting_status": "reported",
          "value": "zero-shot chain-of-thought"
        },
        "repeats": {
          "notes": "Tables 2–4 average three runs per model.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public zero-shot prompt.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "No numeric confidence intervals are reported in Tables 2–4.",
          "reporting_status": "reported",
          "value": "mean across three runs per model"
        },
        "system_prompt_public": {
          "notes": "Appendix D.1 publishes the complete prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No common token budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One multiple-choice completion per question.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibshindprimers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.61
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibshindprimers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.71
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibshindprimers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.85
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibshindprimers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.48
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibshindprimers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.75
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibshindprimers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.65
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibshindprimers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.43
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibshindprimers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.71
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibshindprimers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.61
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibshindprimers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.64
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibshindprimers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.75
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibshindprimers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.85
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibshindprimers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.13
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibshindprimers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.67
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibshindprimers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.19
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibshindprimers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.57
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibshindprimers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.57
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibshindprimers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 1.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibshindprimers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.45
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibshindprimers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.45
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibshindprimers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 1.0
        }
      ],
      "scope": {
        "filter": "Complete creator snapshot, combining public and private splits.",
        "n": 50,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "All registered values were parsed from creator Tables 2–4 and independently checked against the HTML table rows.",
        "status": "verified"
      },
      "work_id": "lab-bench-paper",
      "work_version_id": "lab-bench-paper-2024-07-14"
    },
    {
      "benchmark_id": "lab-bench-seqqa-pcr-gene-gibssmaprimers",
      "benchmark_version": "paper-v3",
      "comparability_group": "lab-bench-seqqa-pcr-gene-gibssmaprimers-paper-v3-zero-shot-cot-no-tools",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-pcr-gene-gibssmaprimers-creator-mcq-protocol",
          "locator": {
            "note": "Full scope, zero-shot chain-of-thought prompt, no tools, three runs, exact metrics, parser, and model identities.",
            "type": "section",
            "value": "Appendix D.1–D.4; Table 1 and Appendix Table 6"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-pcr-gene-gibssmaprimers-creator-mcq-results",
          "locator": {
            "note": "Accuracy, precision, and coverage. Human row excluded.",
            "type": "table",
            "value": "Tables 2–4, row SeqQA_PCR-gene-gibssmaprimers-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-seqqa-pcr-gene-gibssmaprimers-creator-mcq",
      "metrics": [
        {
          "aggregation": "correct / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Accuracy",
          "tolerance": "exact choice",
          "unit": "proportion"
        },
        {
          "aggregation": "correct / attempted questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Precision (selective accuracy)",
          "tolerance": "insufficient-information responses excluded from denominator",
          "unit": "proportion"
        },
        {
          "aggregation": "attempted / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "coverage",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Coverage",
          "tolerance": "insufficient-information responses treated as not attempted",
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "lab-bench-claude-3-5-sonnet-20240620",
        "lab-bench-claude-3-haiku-20240307",
        "lab-bench-claude-3-opus-20240229",
        "lab-bench-gemini-1-5-pro-001",
        "lab-bench-gpt-4-turbo-unversioned",
        "lab-bench-gpt-4o-unversioned",
        "lab-bench-meta-llama-3-70b-instruct"
      ],
      "protocol": {
        "contamination": {
          "notes": "Main results combine both splits.",
          "reporting_status": "reported",
          "value": "full pre-release snapshot with 80% later public and 20% private contamination-monitoring splits"
        },
        "grader": {
          "human_review": false,
          "model": "Claude 2 fallback parser",
          "reporting_status": "reported",
          "type": "regex multiple-choice parser with Claude 2 fallback"
        },
        "reasoning": {
          "notes": "Prompt includes Think step by step.",
          "reporting_status": "reported",
          "value": "zero-shot chain-of-thought"
        },
        "repeats": {
          "notes": "Tables 2–4 average three runs per model.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public zero-shot prompt.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "No numeric confidence intervals are reported in Tables 2–4.",
          "reporting_status": "reported",
          "value": "mean across three runs per model"
        },
        "system_prompt_public": {
          "notes": "Appendix D.1 publishes the complete prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No common token budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One multiple-choice completion per question.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibssmaprimers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.65
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibssmaprimers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.7
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibssmaprimers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.93
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibssmaprimers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.34
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibssmaprimers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.4
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibssmaprimers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.86
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibssmaprimers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.16
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibssmaprimers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.54
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibssmaprimers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.3
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibssmaprimers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.45
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibssmaprimers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.47
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibssmaprimers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.95
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibssmaprimers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.29
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibssmaprimers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.72
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibssmaprimers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.4
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibssmaprimers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.51
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibssmaprimers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.51
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibssmaprimers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.99
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibssmaprimers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.58
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibssmaprimers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.58
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-gene-gibssmaprimers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.99
        }
      ],
      "scope": {
        "filter": "Complete creator snapshot, combining public and private splits.",
        "n": 50,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "All registered values were parsed from creator Tables 2–4 and independently checked against the HTML table rows.",
        "status": "verified"
      },
      "work_id": "lab-bench-paper",
      "work_version_id": "lab-bench-paper-2024-07-14"
    },
    {
      "benchmark_id": "lab-bench-seqqa-pcr-geneprimers-enz",
      "benchmark_version": "paper-v3",
      "comparability_group": "lab-bench-seqqa-pcr-geneprimers-enz-paper-v3-zero-shot-cot-no-tools",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-pcr-geneprimers-enz-creator-mcq-protocol",
          "locator": {
            "note": "Full scope, zero-shot chain-of-thought prompt, no tools, three runs, exact metrics, parser, and model identities.",
            "type": "section",
            "value": "Appendix D.1–D.4; Table 1 and Appendix Table 6"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-pcr-geneprimers-enz-creator-mcq-results",
          "locator": {
            "note": "Accuracy, precision, and coverage. Human row excluded.",
            "type": "table",
            "value": "Tables 2–4, row SeqQA_PCR-geneprimers-enz-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-seqqa-pcr-geneprimers-enz-creator-mcq",
      "metrics": [
        {
          "aggregation": "correct / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Accuracy",
          "tolerance": "exact choice",
          "unit": "proportion"
        },
        {
          "aggregation": "correct / attempted questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Precision (selective accuracy)",
          "tolerance": "insufficient-information responses excluded from denominator",
          "unit": "proportion"
        },
        {
          "aggregation": "attempted / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "coverage",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Coverage",
          "tolerance": "insufficient-information responses treated as not attempted",
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "lab-bench-claude-3-5-sonnet-20240620",
        "lab-bench-claude-3-haiku-20240307",
        "lab-bench-claude-3-opus-20240229",
        "lab-bench-gemini-1-5-pro-001",
        "lab-bench-gpt-4-turbo-unversioned",
        "lab-bench-gpt-4o-unversioned",
        "lab-bench-meta-llama-3-70b-instruct"
      ],
      "protocol": {
        "contamination": {
          "notes": "Main results combine both splits.",
          "reporting_status": "reported",
          "value": "full pre-release snapshot with 80% later public and 20% private contamination-monitoring splits"
        },
        "grader": {
          "human_review": false,
          "model": "Claude 2 fallback parser",
          "reporting_status": "reported",
          "type": "regex multiple-choice parser with Claude 2 fallback"
        },
        "reasoning": {
          "notes": "Prompt includes Think step by step.",
          "reporting_status": "reported",
          "value": "zero-shot chain-of-thought"
        },
        "repeats": {
          "notes": "Tables 2–4 average three runs per model.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public zero-shot prompt.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "No numeric confidence intervals are reported in Tables 2–4.",
          "reporting_status": "reported",
          "value": "mean across three runs per model"
        },
        "system_prompt_public": {
          "notes": "Appendix D.1 publishes the complete prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No common token budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One multiple-choice completion per question.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-geneprimers-enz-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.73
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-geneprimers-enz-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.91
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-geneprimers-enz-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.81
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-geneprimers-enz-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.76
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-geneprimers-enz-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.84
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-geneprimers-enz-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.91
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-geneprimers-enz-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.66
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-geneprimers-enz-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.69
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-geneprimers-enz-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.95
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-geneprimers-enz-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.74
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-geneprimers-enz-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.77
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-geneprimers-enz-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.97
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-geneprimers-enz-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.66
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-geneprimers-enz-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.75
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-geneprimers-enz-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.89
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-geneprimers-enz-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.83
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-geneprimers-enz-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.83
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-geneprimers-enz-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 1.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-geneprimers-enz-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.37
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-geneprimers-enz-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.39
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-geneprimers-enz-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.95
        }
      ],
      "scope": {
        "filter": "Complete creator snapshot, combining public and private splits.",
        "n": 50,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "All registered values were parsed from creator Tables 2–4 and independently checked against the HTML table rows.",
        "status": "verified"
      },
      "work_id": "lab-bench-paper",
      "work_version_id": "lab-bench-paper-2024-07-14"
    },
    {
      "benchmark_id": "lab-bench-seqqa-pcr-len-primers",
      "benchmark_version": "paper-v3",
      "comparability_group": "lab-bench-seqqa-pcr-len-primers-paper-v3-zero-shot-cot-no-tools",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-pcr-len-primers-creator-mcq-protocol",
          "locator": {
            "note": "Full scope, zero-shot chain-of-thought prompt, no tools, three runs, exact metrics, parser, and model identities.",
            "type": "section",
            "value": "Appendix D.1–D.4; Table 1 and Appendix Table 6"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-pcr-len-primers-creator-mcq-results",
          "locator": {
            "note": "Accuracy, precision, and coverage. Human row excluded.",
            "type": "table",
            "value": "Tables 2–4, row SeqQA_PCR-len-primers-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-seqqa-pcr-len-primers-creator-mcq",
      "metrics": [
        {
          "aggregation": "correct / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Accuracy",
          "tolerance": "exact choice",
          "unit": "proportion"
        },
        {
          "aggregation": "correct / attempted questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Precision (selective accuracy)",
          "tolerance": "insufficient-information responses excluded from denominator",
          "unit": "proportion"
        },
        {
          "aggregation": "attempted / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "coverage",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Coverage",
          "tolerance": "insufficient-information responses treated as not attempted",
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "lab-bench-claude-3-5-sonnet-20240620",
        "lab-bench-claude-3-haiku-20240307",
        "lab-bench-claude-3-opus-20240229",
        "lab-bench-gemini-1-5-pro-001",
        "lab-bench-gpt-4-turbo-unversioned",
        "lab-bench-gpt-4o-unversioned",
        "lab-bench-meta-llama-3-70b-instruct"
      ],
      "protocol": {
        "contamination": {
          "notes": "Main results combine both splits.",
          "reporting_status": "reported",
          "value": "full pre-release snapshot with 80% later public and 20% private contamination-monitoring splits"
        },
        "grader": {
          "human_review": false,
          "model": "Claude 2 fallback parser",
          "reporting_status": "reported",
          "type": "regex multiple-choice parser with Claude 2 fallback"
        },
        "reasoning": {
          "notes": "Prompt includes Think step by step.",
          "reporting_status": "reported",
          "value": "zero-shot chain-of-thought"
        },
        "repeats": {
          "notes": "Tables 2–4 average three runs per model.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public zero-shot prompt.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "No numeric confidence intervals are reported in Tables 2–4.",
          "reporting_status": "reported",
          "value": "mean across three runs per model"
        },
        "system_prompt_public": {
          "notes": "Appendix D.1 publishes the complete prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No common token budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One multiple-choice completion per question.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-len-primers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.29
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-len-primers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.29
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-len-primers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.99
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-len-primers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.37
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-len-primers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.45
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-len-primers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.81
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-len-primers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.02
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-len-primers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.07
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-len-primers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.31
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-len-primers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.12
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-len-primers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.14
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-len-primers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.84
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-len-primers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.1
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-len-primers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.13
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-len-primers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.78
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-len-primers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.13
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-len-primers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.14
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-len-primers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.95
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-len-primers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.06
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-len-primers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.06
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-len-primers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 1.0
        }
      ],
      "scope": {
        "filter": "Complete creator snapshot, combining public and private splits.",
        "n": 50,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "All registered values were parsed from creator Tables 2–4 and independently checked against the HTML table rows.",
        "status": "verified"
      },
      "work_id": "lab-bench-paper",
      "work_version_id": "lab-bench-paper-2024-07-14"
    },
    {
      "benchmark_id": "lab-bench-seqqa-pcr-primers-len",
      "benchmark_version": "paper-v3",
      "comparability_group": "lab-bench-seqqa-pcr-primers-len-paper-v3-zero-shot-cot-no-tools",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-pcr-primers-len-creator-mcq-protocol",
          "locator": {
            "note": "Full scope, zero-shot chain-of-thought prompt, no tools, three runs, exact metrics, parser, and model identities.",
            "type": "section",
            "value": "Appendix D.1–D.4; Table 1 and Appendix Table 6"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-pcr-primers-len-creator-mcq-results",
          "locator": {
            "note": "Accuracy, precision, and coverage. Human row excluded.",
            "type": "table",
            "value": "Tables 2–4, row SeqQA_PCR-primers-len-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-seqqa-pcr-primers-len-creator-mcq",
      "metrics": [
        {
          "aggregation": "correct / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Accuracy",
          "tolerance": "exact choice",
          "unit": "proportion"
        },
        {
          "aggregation": "correct / attempted questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Precision (selective accuracy)",
          "tolerance": "insufficient-information responses excluded from denominator",
          "unit": "proportion"
        },
        {
          "aggregation": "attempted / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "coverage",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Coverage",
          "tolerance": "insufficient-information responses treated as not attempted",
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "lab-bench-claude-3-5-sonnet-20240620",
        "lab-bench-claude-3-haiku-20240307",
        "lab-bench-claude-3-opus-20240229",
        "lab-bench-gemini-1-5-pro-001",
        "lab-bench-gpt-4-turbo-unversioned",
        "lab-bench-gpt-4o-unversioned",
        "lab-bench-meta-llama-3-70b-instruct"
      ],
      "protocol": {
        "contamination": {
          "notes": "Main results combine both splits.",
          "reporting_status": "reported",
          "value": "full pre-release snapshot with 80% later public and 20% private contamination-monitoring splits"
        },
        "grader": {
          "human_review": false,
          "model": "Claude 2 fallback parser",
          "reporting_status": "reported",
          "type": "regex multiple-choice parser with Claude 2 fallback"
        },
        "reasoning": {
          "notes": "Prompt includes Think step by step.",
          "reporting_status": "reported",
          "value": "zero-shot chain-of-thought"
        },
        "repeats": {
          "notes": "Tables 2–4 average three runs per model.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public zero-shot prompt.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "No numeric confidence intervals are reported in Tables 2–4.",
          "reporting_status": "reported",
          "value": "mean across three runs per model"
        },
        "system_prompt_public": {
          "notes": "Appendix D.1 publishes the complete prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No common token budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One multiple-choice completion per question.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-primers-len-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.35
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-primers-len-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.37
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-primers-len-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.93
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-primers-len-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.29
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-primers-len-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.38
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-primers-len-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.78
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-primers-len-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.15
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-primers-len-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.19
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-primers-len-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.77
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-primers-len-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.13
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-primers-len-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.15
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-primers-len-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.89
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-primers-len-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.19
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-primers-len-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.26
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-primers-len-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.71
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-primers-len-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.26
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-primers-len-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.26
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-primers-len-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 1.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-primers-len-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.21
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-primers-len-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.22
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-primers-len-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.97
        }
      ],
      "scope": {
        "filter": "Complete creator snapshot, combining public and private splits.",
        "n": 50,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "All registered values were parsed from creator Tables 2–4 and independently checked against the HTML table rows.",
        "status": "verified"
      },
      "work_id": "lab-bench-paper",
      "work_version_id": "lab-bench-paper-2024-07-14"
    },
    {
      "benchmark_id": "lab-bench-seqqa-pcr-seq-enzprimers",
      "benchmark_version": "paper-v3",
      "comparability_group": "lab-bench-seqqa-pcr-seq-enzprimers-paper-v3-zero-shot-cot-no-tools",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-pcr-seq-enzprimers-creator-mcq-protocol",
          "locator": {
            "note": "Full scope, zero-shot chain-of-thought prompt, no tools, three runs, exact metrics, parser, and model identities.",
            "type": "section",
            "value": "Appendix D.1–D.4; Table 1 and Appendix Table 6"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-pcr-seq-enzprimers-creator-mcq-results",
          "locator": {
            "note": "Accuracy, precision, and coverage. Human row excluded.",
            "type": "table",
            "value": "Tables 2–4, row SeqQA_PCR-seq-enzprimers-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-seqqa-pcr-seq-enzprimers-creator-mcq",
      "metrics": [
        {
          "aggregation": "correct / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Accuracy",
          "tolerance": "exact choice",
          "unit": "proportion"
        },
        {
          "aggregation": "correct / attempted questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Precision (selective accuracy)",
          "tolerance": "insufficient-information responses excluded from denominator",
          "unit": "proportion"
        },
        {
          "aggregation": "attempted / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "coverage",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Coverage",
          "tolerance": "insufficient-information responses treated as not attempted",
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "lab-bench-claude-3-5-sonnet-20240620",
        "lab-bench-claude-3-haiku-20240307",
        "lab-bench-claude-3-opus-20240229",
        "lab-bench-gemini-1-5-pro-001",
        "lab-bench-gpt-4-turbo-unversioned",
        "lab-bench-gpt-4o-unversioned",
        "lab-bench-meta-llama-3-70b-instruct"
      ],
      "protocol": {
        "contamination": {
          "notes": "Main results combine both splits.",
          "reporting_status": "reported",
          "value": "full pre-release snapshot with 80% later public and 20% private contamination-monitoring splits"
        },
        "grader": {
          "human_review": false,
          "model": "Claude 2 fallback parser",
          "reporting_status": "reported",
          "type": "regex multiple-choice parser with Claude 2 fallback"
        },
        "reasoning": {
          "notes": "Prompt includes Think step by step.",
          "reporting_status": "reported",
          "value": "zero-shot chain-of-thought"
        },
        "repeats": {
          "notes": "Tables 2–4 average three runs per model.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public zero-shot prompt.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "No numeric confidence intervals are reported in Tables 2–4.",
          "reporting_status": "reported",
          "value": "mean across three runs per model"
        },
        "system_prompt_public": {
          "notes": "Appendix D.1 publishes the complete prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No common token budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One multiple-choice completion per question.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-enzprimers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.94
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-enzprimers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.95
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-enzprimers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.99
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-enzprimers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.8
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-enzprimers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.83
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-enzprimers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.96
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-enzprimers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.54
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-enzprimers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.73
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-enzprimers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.75
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-enzprimers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.83
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-enzprimers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.86
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-enzprimers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.96
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-enzprimers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.67
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-enzprimers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.78
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-enzprimers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.87
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-enzprimers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.86
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-enzprimers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.86
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-enzprimers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 1.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-enzprimers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.73
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-enzprimers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.73
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-enzprimers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 1.0
        }
      ],
      "scope": {
        "filter": "Complete creator snapshot, combining public and private splits.",
        "n": 50,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "All registered values were parsed from creator Tables 2–4 and independently checked against the HTML table rows.",
        "status": "verified"
      },
      "work_id": "lab-bench-paper",
      "work_version_id": "lab-bench-paper-2024-07-14"
    },
    {
      "benchmark_id": "lab-bench-seqqa-pcr-seq-primers",
      "benchmark_version": "paper-v3",
      "comparability_group": "lab-bench-seqqa-pcr-seq-primers-paper-v3-zero-shot-cot-no-tools",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-pcr-seq-primers-creator-mcq-protocol",
          "locator": {
            "note": "Full scope, zero-shot chain-of-thought prompt, no tools, three runs, exact metrics, parser, and model identities.",
            "type": "section",
            "value": "Appendix D.1–D.4; Table 1 and Appendix Table 6"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-pcr-seq-primers-creator-mcq-results",
          "locator": {
            "note": "Accuracy, precision, and coverage. Human row excluded.",
            "type": "table",
            "value": "Tables 2–4, row SeqQA_PCR-seq-primers-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-seqqa-pcr-seq-primers-creator-mcq",
      "metrics": [
        {
          "aggregation": "correct / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Accuracy",
          "tolerance": "exact choice",
          "unit": "proportion"
        },
        {
          "aggregation": "correct / attempted questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Precision (selective accuracy)",
          "tolerance": "insufficient-information responses excluded from denominator",
          "unit": "proportion"
        },
        {
          "aggregation": "attempted / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "coverage",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Coverage",
          "tolerance": "insufficient-information responses treated as not attempted",
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "lab-bench-claude-3-5-sonnet-20240620",
        "lab-bench-claude-3-haiku-20240307",
        "lab-bench-claude-3-opus-20240229",
        "lab-bench-gemini-1-5-pro-001",
        "lab-bench-gpt-4-turbo-unversioned",
        "lab-bench-gpt-4o-unversioned",
        "lab-bench-meta-llama-3-70b-instruct"
      ],
      "protocol": {
        "contamination": {
          "notes": "Main results combine both splits.",
          "reporting_status": "reported",
          "value": "full pre-release snapshot with 80% later public and 20% private contamination-monitoring splits"
        },
        "grader": {
          "human_review": false,
          "model": "Claude 2 fallback parser",
          "reporting_status": "reported",
          "type": "regex multiple-choice parser with Claude 2 fallback"
        },
        "reasoning": {
          "notes": "Prompt includes Think step by step.",
          "reporting_status": "reported",
          "value": "zero-shot chain-of-thought"
        },
        "repeats": {
          "notes": "Tables 2–4 average three runs per model.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public zero-shot prompt.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "No numeric confidence intervals are reported in Tables 2–4.",
          "reporting_status": "reported",
          "value": "mean across three runs per model"
        },
        "system_prompt_public": {
          "notes": "Appendix D.1 publishes the complete prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No common token budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One multiple-choice completion per question.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-primers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.97
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-primers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.98
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-primers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.99
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-primers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.83
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-primers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.91
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-primers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.91
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-primers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.48
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-primers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.64
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-primers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.75
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-primers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.73
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-primers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.76
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-primers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.96
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-primers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.52
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-primers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.56
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-primers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.93
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-primers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.81
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-primers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.82
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-primers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.99
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-primers-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.47
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-primers-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.47
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-pcr-seq-primers-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 1.0
        }
      ],
      "scope": {
        "filter": "Complete creator snapshot, combining public and private splits.",
        "n": 50,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "All registered values were parsed from creator Tables 2–4 and independently checked against the HTML table rows.",
        "status": "verified"
      },
      "work_id": "lab-bench-paper",
      "work_version_id": "lab-bench-paper-2024-07-14"
    },
    {
      "benchmark_id": "lab-bench-seqqa-prop-seq-gcpercent",
      "benchmark_version": "paper-v3",
      "comparability_group": "lab-bench-seqqa-prop-seq-gcpercent-paper-v3-zero-shot-cot-no-tools",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-prop-seq-gcpercent-creator-mcq-protocol",
          "locator": {
            "note": "Full scope, zero-shot chain-of-thought prompt, no tools, three runs, exact metrics, parser, and model identities.",
            "type": "section",
            "value": "Appendix D.1–D.4; Table 1 and Appendix Table 6"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-prop-seq-gcpercent-creator-mcq-results",
          "locator": {
            "note": "Accuracy, precision, and coverage. Human row excluded.",
            "type": "table",
            "value": "Tables 2–4, row SeqQA_Prop-seq-gcpercent-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-seqqa-prop-seq-gcpercent-creator-mcq",
      "metrics": [
        {
          "aggregation": "correct / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Accuracy",
          "tolerance": "exact choice",
          "unit": "proportion"
        },
        {
          "aggregation": "correct / attempted questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Precision (selective accuracy)",
          "tolerance": "insufficient-information responses excluded from denominator",
          "unit": "proportion"
        },
        {
          "aggregation": "attempted / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "coverage",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Coverage",
          "tolerance": "insufficient-information responses treated as not attempted",
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "lab-bench-claude-3-5-sonnet-20240620",
        "lab-bench-claude-3-haiku-20240307",
        "lab-bench-claude-3-opus-20240229",
        "lab-bench-gemini-1-5-pro-001",
        "lab-bench-gpt-4-turbo-unversioned",
        "lab-bench-gpt-4o-unversioned",
        "lab-bench-meta-llama-3-70b-instruct"
      ],
      "protocol": {
        "contamination": {
          "notes": "Main results combine both splits.",
          "reporting_status": "reported",
          "value": "full pre-release snapshot with 80% later public and 20% private contamination-monitoring splits"
        },
        "grader": {
          "human_review": false,
          "model": "Claude 2 fallback parser",
          "reporting_status": "reported",
          "type": "regex multiple-choice parser with Claude 2 fallback"
        },
        "reasoning": {
          "notes": "Prompt includes Think step by step.",
          "reporting_status": "reported",
          "value": "zero-shot chain-of-thought"
        },
        "repeats": {
          "notes": "Tables 2–4 average three runs per model.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public zero-shot prompt.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "No numeric confidence intervals are reported in Tables 2–4.",
          "reporting_status": "reported",
          "value": "mean across three runs per model"
        },
        "system_prompt_public": {
          "notes": "Appendix D.1 publishes the complete prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No common token budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One multiple-choice completion per question.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-prop-seq-gcpercent-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.59
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-prop-seq-gcpercent-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.59
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-prop-seq-gcpercent-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.99
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-prop-seq-gcpercent-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.45
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-prop-seq-gcpercent-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.5
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-prop-seq-gcpercent-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.9
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-prop-seq-gcpercent-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.39
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-prop-seq-gcpercent-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.41
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-prop-seq-gcpercent-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.93
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-prop-seq-gcpercent-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.29
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-prop-seq-gcpercent-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.32
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-prop-seq-gcpercent-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.9
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-prop-seq-gcpercent-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.18
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-prop-seq-gcpercent-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.23
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-prop-seq-gcpercent-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.78
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-prop-seq-gcpercent-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.39
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-prop-seq-gcpercent-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.4
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-prop-seq-gcpercent-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.99
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-prop-seq-gcpercent-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.35
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-prop-seq-gcpercent-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.36
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-prop-seq-gcpercent-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.99
        }
      ],
      "scope": {
        "filter": "Complete creator snapshot, combining public and private splits.",
        "n": 50,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "All registered values were parsed from creator Tables 2–4 and independently checked against the HTML table rows.",
        "status": "verified"
      },
      "work_id": "lab-bench-paper",
      "work_version_id": "lab-bench-paper-2024-07-14"
    },
    {
      "benchmark_id": "lab-bench-seqqa-re-seq-lenfrags",
      "benchmark_version": "paper-v3",
      "comparability_group": "lab-bench-seqqa-re-seq-lenfrags-paper-v3-zero-shot-cot-no-tools",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-re-seq-lenfrags-creator-mcq-protocol",
          "locator": {
            "note": "Full scope, zero-shot chain-of-thought prompt, no tools, three runs, exact metrics, parser, and model identities.",
            "type": "section",
            "value": "Appendix D.1–D.4; Table 1 and Appendix Table 6"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-re-seq-lenfrags-creator-mcq-results",
          "locator": {
            "note": "Accuracy, precision, and coverage. Human row excluded.",
            "type": "table",
            "value": "Tables 2–4, row SeqQA_RE-seq-lenfrags-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-seqqa-re-seq-lenfrags-creator-mcq",
      "metrics": [
        {
          "aggregation": "correct / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Accuracy",
          "tolerance": "exact choice",
          "unit": "proportion"
        },
        {
          "aggregation": "correct / attempted questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Precision (selective accuracy)",
          "tolerance": "insufficient-information responses excluded from denominator",
          "unit": "proportion"
        },
        {
          "aggregation": "attempted / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "coverage",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Coverage",
          "tolerance": "insufficient-information responses treated as not attempted",
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "lab-bench-claude-3-5-sonnet-20240620",
        "lab-bench-claude-3-haiku-20240307",
        "lab-bench-claude-3-opus-20240229",
        "lab-bench-gemini-1-5-pro-001",
        "lab-bench-gpt-4-turbo-unversioned",
        "lab-bench-gpt-4o-unversioned",
        "lab-bench-meta-llama-3-70b-instruct"
      ],
      "protocol": {
        "contamination": {
          "notes": "Main results combine both splits.",
          "reporting_status": "reported",
          "value": "full pre-release snapshot with 80% later public and 20% private contamination-monitoring splits"
        },
        "grader": {
          "human_review": false,
          "model": "Claude 2 fallback parser",
          "reporting_status": "reported",
          "type": "regex multiple-choice parser with Claude 2 fallback"
        },
        "reasoning": {
          "notes": "Prompt includes Think step by step.",
          "reporting_status": "reported",
          "value": "zero-shot chain-of-thought"
        },
        "repeats": {
          "notes": "Tables 2–4 average three runs per model.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public zero-shot prompt.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "No numeric confidence intervals are reported in Tables 2–4.",
          "reporting_status": "reported",
          "value": "mean across three runs per model"
        },
        "system_prompt_public": {
          "notes": "Appendix D.1 publishes the complete prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No common token budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One multiple-choice completion per question.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-lenfrags-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.12
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-lenfrags-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.33
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-lenfrags-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.36
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-lenfrags-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.13
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-lenfrags-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.26
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-lenfrags-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.51
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-lenfrags-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.07
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-lenfrags-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.17
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-lenfrags-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.43
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-lenfrags-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.17
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-lenfrags-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.25
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-lenfrags-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.67
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-lenfrags-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.07
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-lenfrags-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.27
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-lenfrags-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.29
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-lenfrags-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.21
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-lenfrags-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.22
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-lenfrags-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.97
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-lenfrags-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.19
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-lenfrags-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.19
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-lenfrags-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 1.0
        }
      ],
      "scope": {
        "filter": "Complete creator snapshot, combining public and private splits.",
        "n": 50,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "All registered values were parsed from creator Tables 2–4 and independently checked against the HTML table rows.",
        "status": "verified"
      },
      "work_id": "lab-bench-paper",
      "work_version_id": "lab-bench-paper-2024-07-14"
    },
    {
      "benchmark_id": "lab-bench-seqqa-re-seq-numfrags",
      "benchmark_version": "paper-v3",
      "comparability_group": "lab-bench-seqqa-re-seq-numfrags-paper-v3-zero-shot-cot-no-tools",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-re-seq-numfrags-creator-mcq-protocol",
          "locator": {
            "note": "Full scope, zero-shot chain-of-thought prompt, no tools, three runs, exact metrics, parser, and model identities.",
            "type": "section",
            "value": "Appendix D.1–D.4; Table 1 and Appendix Table 6"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-seqqa-re-seq-numfrags-creator-mcq-results",
          "locator": {
            "note": "Accuracy, precision, and coverage. Human row excluded.",
            "type": "table",
            "value": "Tables 2–4, row SeqQA_RE-seq-numfrags-v1"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-seqqa-re-seq-numfrags-creator-mcq",
      "metrics": [
        {
          "aggregation": "correct / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Accuracy",
          "tolerance": "exact choice",
          "unit": "proportion"
        },
        {
          "aggregation": "correct / attempted questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Precision (selective accuracy)",
          "tolerance": "insufficient-information responses excluded from denominator",
          "unit": "proportion"
        },
        {
          "aggregation": "attempted / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "coverage",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Coverage",
          "tolerance": "insufficient-information responses treated as not attempted",
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "lab-bench-claude-3-5-sonnet-20240620",
        "lab-bench-claude-3-haiku-20240307",
        "lab-bench-claude-3-opus-20240229",
        "lab-bench-gemini-1-5-pro-001",
        "lab-bench-gpt-4-turbo-unversioned",
        "lab-bench-gpt-4o-unversioned",
        "lab-bench-meta-llama-3-70b-instruct"
      ],
      "protocol": {
        "contamination": {
          "notes": "Main results combine both splits.",
          "reporting_status": "reported",
          "value": "full pre-release snapshot with 80% later public and 20% private contamination-monitoring splits"
        },
        "grader": {
          "human_review": false,
          "model": "Claude 2 fallback parser",
          "reporting_status": "reported",
          "type": "regex multiple-choice parser with Claude 2 fallback"
        },
        "reasoning": {
          "notes": "Prompt includes Think step by step.",
          "reporting_status": "reported",
          "value": "zero-shot chain-of-thought"
        },
        "repeats": {
          "notes": "Tables 2–4 average three runs per model.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public zero-shot prompt.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "No numeric confidence intervals are reported in Tables 2–4.",
          "reporting_status": "reported",
          "value": "mean across three runs per model"
        },
        "system_prompt_public": {
          "notes": "Appendix D.1 publishes the complete prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No common token budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One multiple-choice completion per question.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-numfrags-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.1
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-numfrags-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.28
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-numfrags-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.35
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-numfrags-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.13
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-numfrags-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.18
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-numfrags-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.76
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-numfrags-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.17
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-numfrags-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.22
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-numfrags-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.75
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-numfrags-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.09
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-numfrags-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.11
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-numfrags-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.81
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-numfrags-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.12
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-numfrags-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.22
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-numfrags-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.54
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-numfrags-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.12
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-numfrags-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.12
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-numfrags-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.97
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-numfrags-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.15
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-numfrags-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.16
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-seqqa-re-seq-numfrags-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 50,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.96
        }
      ],
      "scope": {
        "filter": "Complete creator snapshot, combining public and private splits.",
        "n": 50,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "All registered values were parsed from creator Tables 2–4 and independently checked against the HTML table rows.",
        "status": "verified"
      },
      "work_id": "lab-bench-paper",
      "work_version_id": "lab-bench-paper-2024-07-14"
    },
    {
      "benchmark_id": "lab-bench-suppqa",
      "benchmark_version": "paper-v3",
      "comparability_group": "lab-bench-suppqa-paper-v3-zero-shot-cot-no-tools",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-suppqa-creator-mcq-protocol",
          "locator": {
            "note": "Full scope, zero-shot chain-of-thought prompt, no tools, three runs, exact metrics, parser, and model identities.",
            "type": "section",
            "value": "Appendix D.1–D.4; Table 1 and Appendix Table 6"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-suppqa-creator-mcq-results",
          "locator": {
            "note": "Accuracy, precision, and coverage. Human row excluded.",
            "type": "table",
            "value": "Tables 2–4, row SuppQA"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-suppqa-creator-mcq",
      "metrics": [
        {
          "aggregation": "correct / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Accuracy",
          "tolerance": "exact choice",
          "unit": "proportion"
        },
        {
          "aggregation": "correct / attempted questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Precision (selective accuracy)",
          "tolerance": "insufficient-information responses excluded from denominator",
          "unit": "proportion"
        },
        {
          "aggregation": "attempted / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "coverage",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Coverage",
          "tolerance": "insufficient-information responses treated as not attempted",
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "lab-bench-claude-3-5-sonnet-20240620",
        "lab-bench-claude-3-haiku-20240307",
        "lab-bench-claude-3-opus-20240229",
        "lab-bench-gemini-1-5-pro-001",
        "lab-bench-gpt-4-turbo-unversioned",
        "lab-bench-gpt-4o-unversioned",
        "lab-bench-meta-llama-3-70b-instruct"
      ],
      "protocol": {
        "contamination": {
          "notes": "Main results combine both splits.",
          "reporting_status": "reported",
          "value": "full pre-release snapshot with 80% later public and 20% private contamination-monitoring splits"
        },
        "grader": {
          "human_review": false,
          "model": "Claude 2 fallback parser",
          "reporting_status": "reported",
          "type": "regex multiple-choice parser with Claude 2 fallback"
        },
        "reasoning": {
          "notes": "Prompt includes Think step by step.",
          "reporting_status": "reported",
          "value": "zero-shot chain-of-thought"
        },
        "repeats": {
          "notes": "Tables 2–4 average three runs per model.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public zero-shot prompt.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "No numeric confidence intervals are reported in Tables 2–4.",
          "reporting_status": "reported",
          "value": "mean across three runs per model"
        },
        "system_prompt_public": {
          "notes": "Appendix D.1 publishes the complete prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No common token budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One multiple-choice completion per question.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-suppqa-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 102,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.02
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-suppqa-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 102,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.75
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-suppqa-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 102,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.04
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-suppqa-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 102,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.02
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-suppqa-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 102,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.59
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-suppqa-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 102,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.04
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-suppqa-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 102,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.01
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-suppqa-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 102,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.32
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-suppqa-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 102,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.04
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-suppqa-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 102,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.13
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-suppqa-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 102,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.47
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-suppqa-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 102,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.29
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-suppqa-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 102,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.06
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-suppqa-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 102,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.39
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-suppqa-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 102,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.14
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-suppqa-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 102,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.18
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-suppqa-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 102,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.35
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-suppqa-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 102,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.53
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-suppqa-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 102,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.2
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-suppqa-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 102,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.27
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-suppqa-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-meta-llama-3-70b-instruct",
          "n": 102,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.74
        }
      ],
      "scope": {
        "filter": "Complete creator snapshot, combining public and private splits.",
        "n": 102,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "All registered values were parsed from creator Tables 2–4 and independently checked against the HTML table rows.",
        "status": "verified"
      },
      "work_id": "lab-bench-paper",
      "work_version_id": "lab-bench-paper-2024-07-14"
    },
    {
      "benchmark_id": "lab-bench-tableqa",
      "benchmark_version": "paper-v3",
      "comparability_group": "lab-bench-tableqa-paper-v3-zero-shot-cot-no-tools",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-tableqa-creator-mcq-protocol",
          "locator": {
            "note": "Full scope, zero-shot chain-of-thought prompt, no tools, three runs, exact metrics, parser, and model identities.",
            "type": "section",
            "value": "Appendix D.1–D.4; Table 1 and Appendix Table 6"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lab-bench-tableqa-creator-mcq-results",
          "locator": {
            "note": "Accuracy, precision, and coverage. Human row excluded.",
            "type": "table",
            "value": "Tables 2–4, row TableQA"
          },
          "source_id": "lab-bench-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lab-bench-tableqa-creator-mcq",
      "metrics": [
        {
          "aggregation": "correct / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Accuracy",
          "tolerance": "exact choice",
          "unit": "proportion"
        },
        {
          "aggregation": "correct / attempted questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Precision (selective accuracy)",
          "tolerance": "insufficient-information responses excluded from denominator",
          "unit": "proportion"
        },
        {
          "aggregation": "attempted / all questions; mean over three runs",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "coverage",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Coverage",
          "tolerance": "insufficient-information responses treated as not attempted",
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "lab-bench-claude-3-5-sonnet-20240620",
        "lab-bench-claude-3-haiku-20240307",
        "lab-bench-claude-3-opus-20240229",
        "lab-bench-gemini-1-5-pro-001",
        "lab-bench-gpt-4-turbo-unversioned",
        "lab-bench-gpt-4o-unversioned"
      ],
      "protocol": {
        "contamination": {
          "notes": "Main results combine both splits.",
          "reporting_status": "reported",
          "value": "full pre-release snapshot with 80% later public and 20% private contamination-monitoring splits"
        },
        "grader": {
          "human_review": false,
          "model": "Claude 2 fallback parser",
          "reporting_status": "reported",
          "type": "regex multiple-choice parser with Claude 2 fallback"
        },
        "reasoning": {
          "notes": "Prompt includes Think step by step.",
          "reporting_status": "reported",
          "value": "zero-shot chain-of-thought"
        },
        "repeats": {
          "notes": "Tables 2–4 average three runs per model.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Public zero-shot prompt.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "No numeric confidence intervals are reported in Tables 2–4.",
          "reporting_status": "reported",
          "value": "mean across three runs per model"
        },
        "system_prompt_public": {
          "notes": "Appendix D.1 publishes the complete prompt template.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No common token budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "Creator baselines used no tools, including no internet.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One multiple-choice completion per question.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-tableqa-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 305,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.83
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-tableqa-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 305,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.9
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-tableqa-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-5-sonnet-20240620",
          "n": 305,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.92
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-tableqa-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 305,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.67
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-tableqa-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 305,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.74
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-tableqa-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-opus-20240229",
          "n": 305,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.9
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-tableqa-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 305,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.59
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-tableqa-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 305,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.71
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-tableqa-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gemini-1-5-pro-001",
          "n": 305,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.83
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-tableqa-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 305,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.71
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-tableqa-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 305,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.75
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-tableqa-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4o-unversioned",
          "n": 305,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.95
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-tableqa-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 305,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.54
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-tableqa-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 305,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.58
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-tableqa-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-gpt-4-turbo-unversioned",
          "n": 305,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.93
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-tableqa-creator-mcq-results"
          ],
          "metric_id": "accuracy",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 305,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.49
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-tableqa-creator-mcq-results"
          ],
          "metric_id": "precision",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 305,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.51
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lab-bench-tableqa-creator-mcq-results"
          ],
          "metric_id": "coverage",
          "model_id": "lab-bench-claude-3-haiku-20240307",
          "n": 305,
          "notes": "Creator-paper full-snapshot result; human-baseline row is not encoded as a model result.",
          "status": "verified",
          "value": 0.96
        }
      ],
      "scope": {
        "filter": "Complete creator snapshot, combining public and private splits.",
        "n": 305,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "All registered values were parsed from creator Tables 2–4 and independently checked against the HTML table rows.",
        "status": "verified"
      },
      "work_id": "lab-bench-paper",
      "work_version_id": "lab-bench-paper-2024-07-14"
    },
    {
      "benchmark_id": "lifescibench",
      "benchmark_version": "initial-release",
      "comparability_group": "lifescibench-initial-release-full-official",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "lifescibench-evidence-run-scope",
          "locator": {
            "note": "States single-turn evaluation, unrestricted Internet browsing, and evaluation of five models across all 750 questions.",
            "type": "section",
            "value": "pp. 7 and 15, Sections 5.1 and 8"
          },
          "source_id": "lifescibench-preprint",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/protocol/turns",
            "/protocol/tools/browser",
            "/protocol/tools/internet"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lifescibench-evidence-metrics",
          "locator": {
            "note": "Defines rubric grading, normalized score, 70% task threshold, problem weighting, automated/model-assisted grading, and expert spot-validation.",
            "type": "section",
            "value": "pp. 7–8, Sections 5.1–5.3"
          },
          "source_id": "lifescibench-preprint",
          "source_type": "work",
          "supports": [
            "/protocol/grader",
            "/protocol/statistical",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lifescibench-evidence-unreported-settings",
          "locator": {
            "note": "Complete public protocol and disclosure sections do not state shots, system prompt, model effort, non-browser tools, budgets, temperature, seed, repeat count, or contamination analysis.",
            "type": "section",
            "value": "pp. 7–8 and 16, Sections 5.1–5.3 and Appendix A"
          },
          "source_id": "lifescibench-preprint",
          "source_type": "work",
          "supports": [
            "/protocol/shots",
            "/protocol/system_prompt_public",
            "/protocol/reasoning",
            "/protocol/tools/databases",
            "/protocol/tools/code_execution",
            "/protocol/tools/container",
            "/protocol/tools/external_tools",
            "/protocol/token_budget",
            "/protocol/time_budget",
            "/protocol/temperature",
            "/protocol/seed",
            "/protocol/repeats",
            "/protocol/contamination"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "lifescibench-evidence-results",
          "locator": {
            "note": "Reports overall normalized score and task pass rate for all five models.",
            "type": "section",
            "value": "p. 9, Section 6.1 and Figure 4"
          },
          "source_id": "lifescibench-preprint",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "lifescibench-official-full",
      "metrics": [
        {
          "aggregation": "problem-weighted mean with each task weighted equally",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "rubric-score",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Normalized rubric score",
          "tolerance": null,
          "unit": "proportion"
        },
        {
          "aggregation": "fraction of tasks whose normalized rubric score is at least 70%",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "pass-rate",
          "pass_threshold": 70,
          "range": [
            0,
            100
          ],
          "source_label": "Task pass rate",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [
        "gemini-3-1-pro",
        "gpt-5-4",
        "gpt-5-5",
        "gpt-rosalind",
        "grok-4-3"
      ],
      "protocol": {
        "contamination": {
          "notes": "The report describes release restrictions but does not report a contamination or decontamination analysis.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": true,
          "model": null,
          "reporting_status": "reported",
          "type": "task-specific expert rubric; automated or model-assisted where used; expert spot-validation on a stratified response subset"
        },
        "reasoning": {
          "notes": "Models were asked to include reasoning, calculations, caveats, or assumptions when useful, but model-specific reasoning/effort settings are not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "repeats": {
          "notes": "The number of model samples or repeated episodes per task is not reported; it must not be inferred as one.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not specified in the official report.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Not specified in the official report.",
          "reporting_status": "not_reported",
          "value": null
        },
        "statistical": {
          "notes": "Subgroup metrics use the same definitions unless otherwise specified; no confidence intervals are reported for overall results.",
          "reporting_status": "reported",
          "value": "problem-weighted mean with each task weighted equally"
        },
        "system_prompt_public": {
          "notes": "Not specified in the official report.",
          "reporting_status": "not_reported",
          "value": null
        },
        "temperature": {
          "notes": "Not specified in the official report.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not specified in the official report.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not specified in the official report.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Unrestricted Internet browsing was permitted.",
            "reporting_status": "reported",
            "value": true
          },
          "code_execution": {
            "notes": "Not specified in the official report.",
            "reporting_status": "not_reported",
            "value": null
          },
          "container": {
            "notes": "Not specified in the official report.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Not specified in the official report.",
            "reporting_status": "not_reported",
            "value": null
          },
          "external_tools": {
            "notes": "Not specified in the official report.",
            "reporting_status": "not_reported",
            "value": null
          },
          "internet": {
            "notes": "Unrestricted Internet browsing was permitted.",
            "reporting_status": "reported",
            "value": true
          }
        },
        "turns": {
          "notes": "Each model receives the prompt and associated artifacts once and produces a final answer without follow-up interaction.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lifescibench-evidence-results"
          ],
          "metric_id": "rubric-score",
          "model_id": "gpt-rosalind",
          "n": 750,
          "notes": "n is the number of tasks; repeats are not reported.",
          "status": "verified",
          "value": 0.576
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lifescibench-evidence-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-rosalind",
          "n": 750,
          "notes": "n is the number of tasks; repeats are not reported.",
          "status": "verified",
          "value": 36.1
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lifescibench-evidence-results"
          ],
          "metric_id": "rubric-score",
          "model_id": "gpt-5-5",
          "n": 750,
          "notes": "n is the number of tasks; repeats are not reported.",
          "status": "verified",
          "value": 0.519
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lifescibench-evidence-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-5",
          "n": 750,
          "notes": "n is the number of tasks; repeats are not reported.",
          "status": "verified",
          "value": 25.7
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lifescibench-evidence-results"
          ],
          "metric_id": "rubric-score",
          "model_id": "gemini-3-1-pro",
          "n": 750,
          "notes": "n is the number of tasks; repeats are not reported.",
          "status": "verified",
          "value": 0.515
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lifescibench-evidence-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gemini-3-1-pro",
          "n": 750,
          "notes": "n is the number of tasks; repeats are not reported.",
          "status": "verified",
          "value": 23.6
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lifescibench-evidence-results"
          ],
          "metric_id": "rubric-score",
          "model_id": "gpt-5-4",
          "n": 750,
          "notes": "n is the number of tasks; repeats are not reported.",
          "status": "verified",
          "value": 0.479
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lifescibench-evidence-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "gpt-5-4",
          "n": 750,
          "notes": "n is the number of tasks; repeats are not reported.",
          "status": "verified",
          "value": 20.7
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lifescibench-evidence-results"
          ],
          "metric_id": "rubric-score",
          "model_id": "grok-4-3",
          "n": 750,
          "notes": "n is the number of tasks; repeats are not reported.",
          "status": "verified",
          "value": 0.399
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "lifescibench-evidence-results"
          ],
          "metric_id": "pass-rate",
          "model_id": "grok-4-3",
          "n": 750,
          "notes": "n is the number of tasks; repeats are not reported.",
          "status": "verified",
          "value": 13.0
        }
      ],
      "scope": {
        "filter": null,
        "n": 750,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Full-scope protocol, metrics, and values checked against Sections 5–6 of the official report; unreported settings remain null.",
        "status": "verified"
      },
      "work_id": "lifescibench-preprint",
      "work_version_id": "lifescibench-preprint-2026-06-17"
    },
    {
      "benchmark_id": "moleculenet",
      "benchmark_version": "original-2017",
      "comparability_group": "moleculenet-original-task-native",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "moleculenet-creator-full-protocol-evidence",
          "locator": {
            "note": "Defines all datasets, 80/10/10 partitions, recommended split/metric per collection, featurizers, models, three-run mean/standard-deviation aggregation, and creator results.",
            "type": "table",
            "value": "Methods Sections 3.1-3.5; Results and Discussion; Appendix Performances; Tables 1-3"
          },
          "source_id": "moleculenet-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/protocol",
            "/metrics"
          ]
        }
      ],
      "id": "moleculenet-creator-full",
      "metrics": [
        {
          "aggregation": "held-out examples and endpoints",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "mean-absolute-error",
          "pass_threshold": null,
          "range": null,
          "source_label": "MAE",
          "tolerance": null,
          "unit": "dataset-specific property units"
        },
        {
          "aggregation": "held-out examples",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "root-mean-squared-error",
          "pass_threshold": null,
          "range": null,
          "source_label": "RMSE",
          "tolerance": null,
          "unit": "dataset-specific property units"
        },
        {
          "aggregation": "dataset-specific macro endpoint average where applicable",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "roc-auc",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "ROC-AUC",
          "tolerance": null,
          "unit": "area"
        },
        {
          "aggregation": "dataset-specific endpoint average where applicable",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "prc-auc",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "PRC-AUC",
          "tolerance": null,
          "unit": "area"
        }
      ],
      "model_ids": [],
      "protocol": {
        "contamination": {
          "notes": "The paper warns random splits can be misleading for chemical series.",
          "reporting_status": "reported",
          "value": "80/10/10 train-validation-test partitions with dataset-specific random, stratified, scaffold, or time splits."
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "deterministic dataset-specific scorer"
        },
        "reasoning": {
          "notes": "Supervised molecular machine learning; no prompting.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "repeats": {
          "notes": "Final benchmark performances use three independent runs with different fixed numerical seeds.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "No single common seed is reported for all benchmark runs.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Supervised molecular machine learning; no prompting.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "statistical": {
          "notes": "Results must preserve dataset, split, featurizer, and model identity.",
          "reporting_status": "reported",
          "value": "Mean and standard deviation over three independent runs for each dataset-model setting; no cross-dataset normalized total."
        },
        "system_prompt_public": {
          "notes": "Supervised molecular machine learning; no prompting.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "temperature": {
          "notes": "Supervised molecular machine learning; no prompting.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "time_budget": {
          "notes": "No common runtime budget across model classes is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Supervised molecular machine learning; no prompting.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Supervised molecular machine learning; no prompting.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "code_execution": {
            "notes": "DeepChem training and evaluation are the benchmark procedure.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "container": {
            "notes": "No single common container is reported in the paper.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Supervised molecular machine learning; no prompting.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "external_tools": {
            "notes": "Dataset-specific recommended splits and metrics are mandatory for comparability.",
            "reporting_status": "reported",
            "value": "DeepChem featurizers, splitters, conventional ML and graph-based models"
          },
          "internet": {
            "notes": "Supervised molecular machine learning; no prompting.",
            "reporting_status": "not_applicable",
            "value": null
          }
        },
        "turns": {
          "notes": "Supervised molecular machine learning; no prompting.",
          "reporting_status": "not_applicable",
          "value": null
        }
      },
      "results": [],
      "scope": {
        "filter": "All seventeen original-paper dataset collections.",
        "n": 17,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Original 17-collection scope, recommended splits, three-run aggregation, and native metrics are verified; no global MoleculeNet score is created.",
        "status": "verified"
      },
      "work_id": "moleculenet-paper",
      "work_version_id": "moleculenet-paper-2017-10-31"
    },
    {
      "benchmark_id": "proteingym-dms-substitutions",
      "benchmark_version": "1.0",
      "comparability_group": "proteingym-v10-dms-substitutions-zero-shot",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "proteingym-eval-scope-protocol",
          "locator": {
            "note": "Defines v1.0, 217 substitution assays, zero-shot label access, baselines, and official scoring.",
            "type": "section",
            "value": "pp. 5–8 and 29, Table 1, Sections 3.3–4.1, and Table A1"
          },
          "source_id": "proteingym-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/protocol"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "proteingym-eval-metrics",
          "locator": {
            "note": "Defines Spearman, AUC, MCC, NDCG@10%, top-10% recall, corrected-average aggregation, and 10,000-sample bootstrap comparisons.",
            "type": "table",
            "value": "pp. 3, 6, 9, 35, and 38; Figure 1, Section 4.1, Table 2, Appendix A.5.1, and Table A5"
          },
          "source_id": "proteingym-paper",
          "source_type": "work",
          "supports": [
            "/metrics",
            "/protocol/statistical"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "proteingym-eval-results",
          "locator": {
            "note": "All 50 zero-shot DMS substitution Spearman values and bootstrap standard errors of model-to-best differences.",
            "type": "table",
            "value": "p. 38, Table A5"
          },
          "source_id": "proteingym-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "proteingym-v10-dms-substitutions-zero-shot",
      "metrics": [
        {
          "aggregation": "assay-level Spearman, averaged within five functional groups, then equal-weight mean across groups",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "spearman-correlation",
          "pass_threshold": null,
          "range": [
            -1,
            1
          ],
          "source_label": "Spearman",
          "tolerance": null,
          "unit": "correlation"
        },
        {
          "aggregation": "assay-level AUC, averaged within five functional groups, then equal-weight mean across groups",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "auc-roc",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "AUC",
          "tolerance": null,
          "unit": "area under ROC curve"
        },
        {
          "aggregation": "assay-level MCC, averaged within five functional groups, then equal-weight mean across groups",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "matthews-correlation",
          "pass_threshold": null,
          "range": [
            -1,
            1
          ],
          "source_label": "MCC",
          "tolerance": null,
          "unit": "correlation"
        },
        {
          "aggregation": "assay-level NDCG@10%, averaged within five functional groups, then equal-weight mean across groups",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "ndcg-at-10-percent",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "NDCG@10%",
          "tolerance": null,
          "unit": "normalized discounted cumulative gain"
        },
        {
          "aggregation": "assay-level top-10% recall, averaged within five functional groups, then equal-weight mean across groups",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "top-10-percent-recall",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Top 10% recall",
          "tolerance": null,
          "unit": "proportion"
        }
      ],
      "model_ids": [
        "carp-38m",
        "carp-600k",
        "carp-640m",
        "carp-76m",
        "deepsequence-ensemble",
        "deepsequence-single",
        "esm-1b",
        "esm-1v-ensemble",
        "esm-1v-single",
        "esm-if1",
        "esm2-150m",
        "esm2-15b",
        "esm2-35m",
        "esm2-3b",
        "esm2-650m",
        "esm2-8m",
        "eve-ensemble",
        "eve-single",
        "evmutation",
        "gemme",
        "mif",
        "mif-st",
        "msa-transformer-ensemble",
        "msa-transformer-single",
        "progen2-base",
        "progen2-l",
        "progen2-m",
        "progen2-s",
        "progen2-xl",
        "proteinmpnn",
        "protgpt2",
        "rita-l",
        "rita-m",
        "rita-s",
        "rita-xl",
        "site-independent",
        "trancepteve-l",
        "trancepteve-m",
        "trancepteve-s",
        "tranception-l",
        "tranception-l-no-retrieval",
        "tranception-m",
        "tranception-m-no-retrieval",
        "tranception-s",
        "tranception-s-no-retrieval",
        "unirep",
        "unirep-evotuned",
        "vespa",
        "vespal",
        "wavenet"
      ],
      "protocol": {
        "contamination": {
          "notes": "Model-specific training overlap is not normalized in the creator-paper result table.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "deterministic official scoring pipeline"
        },
        "reasoning": {
          "notes": "No natural-language prompt.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "repeats": {
          "notes": "Repeated inference count is not reported as a common protocol field.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Baseline-specific stochastic settings are not normalized in the aggregate result table.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Here zero-shot describes access to assay labels, not natural-language in-context examples.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "Table A5 reports standard error of the difference from TranceptEVE, not a confidence interval for each raw score.",
          "reporting_status": "reported",
          "value": "Non-parametric bootstrap standard error of each model-to-best Spearman difference over 10,000 bootstrap samples from proteins"
        },
        "system_prompt_public": {
          "notes": "No natural-language prompt.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "temperature": {
          "notes": "No natural-language prompt.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "time_budget": {
          "notes": "No common runtime budget is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "No natural-language prompt.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "No natural-language prompt.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "code_execution": {
            "notes": "Inference is performed by model-specific code; code execution is not an affordance granted to an evaluated agent.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "container": {
            "notes": "The paper does not standardize one container across baselines.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "No natural-language prompt.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "external_tools": {
            "notes": "Baselines use their defined input modalities, including single sequences, MSAs, or structures; this is not a common tool-use condition.",
            "reporting_status": "reported",
            "value": "model-specific inputs"
          },
          "internet": {
            "notes": "No natural-language prompt.",
            "reporting_status": "not_applicable",
            "value": null
          }
        },
        "turns": {
          "notes": "Batch mutation-effect prediction benchmark.",
          "reporting_status": "not_applicable",
          "value": "not-applicable"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "trancepteve-l",
          "n": null,
          "notes": "Table A5 rank 1*; bootstrap SE of difference from the best model: 0.000.",
          "status": "verified",
          "value": 0.456
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "trancepteve-m",
          "n": null,
          "notes": "Table A5 rank 1*; bootstrap SE of difference from the best model: 0.004.",
          "status": "verified",
          "value": 0.455
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "gemme",
          "n": null,
          "notes": "Table A5 rank 1*; bootstrap SE of difference from the best model: 0.007.",
          "status": "verified",
          "value": 0.455
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "trancepteve-s",
          "n": null,
          "notes": "Table A5 rank 4; bootstrap SE of difference from the best model: 0.004.",
          "status": "verified",
          "value": 0.452
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "eve-ensemble",
          "n": null,
          "notes": "Table A5 rank 5; bootstrap SE of difference from the best model: 0.006.",
          "status": "verified",
          "value": 0.439
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "vespa",
          "n": null,
          "notes": "Table A5 rank 6; bootstrap SE of difference from the best model: 0.006.",
          "status": "verified",
          "value": 0.436
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "tranception-l",
          "n": null,
          "notes": "Table A5 rank 7*; bootstrap SE of difference from the best model: 0.004.",
          "status": "verified",
          "value": 0.434
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "msa-transformer-ensemble",
          "n": null,
          "notes": "Table A5 rank 7*; bootstrap SE of difference from the best model: 0.009.",
          "status": "verified",
          "value": 0.434
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "eve-single",
          "n": null,
          "notes": "Table A5 rank 9; bootstrap SE of difference from the best model: 0.005.",
          "status": "verified",
          "value": 0.433
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "tranception-m",
          "n": null,
          "notes": "Table A5 rank 10; bootstrap SE of difference from the best model: 0.005.",
          "status": "verified",
          "value": 0.427
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "esm-if1",
          "n": null,
          "notes": "Table A5 rank 11; bootstrap SE of difference from the best model: 0.011.",
          "status": "verified",
          "value": 0.422
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "msa-transformer-single",
          "n": null,
          "notes": "Table A5 rank 12; bootstrap SE of difference from the best model: 0.009.",
          "status": "verified",
          "value": 0.421
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "deepsequence-ensemble",
          "n": null,
          "notes": "Table A5 rank 13; bootstrap SE of difference from the best model: 0.008.",
          "status": "verified",
          "value": 0.419
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "tranception-s",
          "n": null,
          "notes": "Table A5 rank 14; bootstrap SE of difference from the best model: 0.006.",
          "status": "verified",
          "value": 0.418
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "esm2-650m",
          "n": null,
          "notes": "Table A5 rank 15; bootstrap SE of difference from the best model: 0.012.",
          "status": "verified",
          "value": 0.414
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "deepsequence-single",
          "n": null,
          "notes": "Table A5 rank 16*; bootstrap SE of difference from the best model: 0.008.",
          "status": "verified",
          "value": 0.407
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "esm-1v-ensemble",
          "n": null,
          "notes": "Table A5 rank 16*; bootstrap SE of difference from the best model: 0.012.",
          "status": "verified",
          "value": 0.407
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "esm2-3b",
          "n": null,
          "notes": "Table A5 rank 18; bootstrap SE of difference from the best model: 0.011.",
          "status": "verified",
          "value": 0.406
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "mif-st",
          "n": null,
          "notes": "Table A5 rank 19*; bootstrap SE of difference from the best model: 0.010.",
          "status": "verified",
          "value": 0.401
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "esm2-15b",
          "n": null,
          "notes": "Table A5 rank 19*; bootstrap SE of difference from the best model: 0.010.",
          "status": "verified",
          "value": 0.401
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "evmutation",
          "n": null,
          "notes": "Table A5 rank 21; bootstrap SE of difference from the best model: 0.006.",
          "status": "verified",
          "value": 0.395
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "esm-1b",
          "n": null,
          "notes": "Table A5 rank 22*; bootstrap SE of difference from the best model: 0.010.",
          "status": "verified",
          "value": 0.394
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "vespal",
          "n": null,
          "notes": "Table A5 rank 22*; bootstrap SE of difference from the best model: 0.007.",
          "status": "verified",
          "value": 0.394
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "progen2-xl",
          "n": null,
          "notes": "Table A5 rank 24; bootstrap SE of difference from the best model: 0.008.",
          "status": "verified",
          "value": 0.391
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "esm2-150m",
          "n": null,
          "notes": "Table A5 rank 25; bootstrap SE of difference from the best model: 0.013.",
          "status": "verified",
          "value": 0.387
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "mif",
          "n": null,
          "notes": "Table A5 rank 26; bootstrap SE of difference from the best model: 0.011.",
          "status": "verified",
          "value": 0.382
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "progen2-l",
          "n": null,
          "notes": "Table A5 rank 27; bootstrap SE of difference from the best model: 0.008.",
          "status": "verified",
          "value": 0.38
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "progen2-m",
          "n": null,
          "notes": "Table A5 rank 28; bootstrap SE of difference from the best model: 0.008.",
          "status": "verified",
          "value": 0.379
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "progen2-base",
          "n": null,
          "notes": "Table A5 rank 29; bootstrap SE of difference from the best model: 0.009.",
          "status": "verified",
          "value": 0.378
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "tranception-l-no-retrieval",
          "n": null,
          "notes": "Table A5 rank 30*; bootstrap SE of difference from the best model: 0.008.",
          "status": "verified",
          "value": 0.374
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "esm-1v-single",
          "n": null,
          "notes": "Table A5 rank 30*; bootstrap SE of difference from the best model: 0.013.",
          "status": "verified",
          "value": 0.374
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "wavenet",
          "n": null,
          "notes": "Table A5 rank 32; bootstrap SE of difference from the best model: 0.012.",
          "status": "verified",
          "value": 0.373
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "rita-xl",
          "n": null,
          "notes": "Table A5 rank 33; bootstrap SE of difference from the best model: 0.009.",
          "status": "verified",
          "value": 0.372
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "carp-640m",
          "n": null,
          "notes": "Table A5 rank 34; bootstrap SE of difference from the best model: 0.011.",
          "status": "verified",
          "value": 0.368
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "rita-l",
          "n": null,
          "notes": "Table A5 rank 35; bootstrap SE of difference from the best model: 0.009.",
          "status": "verified",
          "value": 0.365
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "site-independent",
          "n": null,
          "notes": "Table A5 rank 36; bootstrap SE of difference from the best model: 0.010.",
          "status": "verified",
          "value": 0.359
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "rita-m",
          "n": null,
          "notes": "Table A5 rank 37; bootstrap SE of difference from the best model: 0.010.",
          "status": "verified",
          "value": 0.35
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "tranception-m-no-retrieval",
          "n": null,
          "notes": "Table A5 rank 38; bootstrap SE of difference from the best model: 0.009.",
          "status": "verified",
          "value": 0.348
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "unirep-evotuned",
          "n": null,
          "notes": "Table A5 rank 39; bootstrap SE of difference from the best model: 0.009.",
          "status": "verified",
          "value": 0.347
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "progen2-s",
          "n": null,
          "notes": "Table A5 rank 40; bootstrap SE of difference from the best model: 0.012.",
          "status": "verified",
          "value": 0.336
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "carp-76m",
          "n": null,
          "notes": "Table A5 rank 41; bootstrap SE of difference from the best model: 0.012.",
          "status": "verified",
          "value": 0.328
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "esm2-35m",
          "n": null,
          "notes": "Table A5 rank 42; bootstrap SE of difference from the best model: 0.015.",
          "status": "verified",
          "value": 0.321
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "rita-s",
          "n": null,
          "notes": "Table A5 rank 43; bootstrap SE of difference from the best model: 0.011.",
          "status": "verified",
          "value": 0.304
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "tranception-s-no-retrieval",
          "n": null,
          "notes": "Table A5 rank 44; bootstrap SE of difference from the best model: 0.012.",
          "status": "verified",
          "value": 0.303
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "carp-38m",
          "n": null,
          "notes": "Table A5 rank 45; bootstrap SE of difference from the best model: 0.014.",
          "status": "verified",
          "value": 0.279
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "proteinmpnn",
          "n": null,
          "notes": "Table A5 rank 46; bootstrap SE of difference from the best model: 0.011.",
          "status": "verified",
          "value": 0.258
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "esm2-8m",
          "n": null,
          "notes": "Table A5 rank 47; bootstrap SE of difference from the best model: 0.015.",
          "status": "verified",
          "value": 0.226
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "unirep",
          "n": null,
          "notes": "Table A5 rank 48; bootstrap SE of difference from the best model: 0.016.",
          "status": "verified",
          "value": 0.19
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "protgpt2",
          "n": null,
          "notes": "Table A5 rank 49; bootstrap SE of difference from the best model: 0.011.",
          "status": "verified",
          "value": 0.188
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteingym-eval-results"
          ],
          "metric_id": "spearman-correlation",
          "model_id": "carp-600k",
          "n": null,
          "notes": "Table A5 rank 50; bootstrap SE of difference from the best model: 0.016.",
          "status": "verified",
          "value": 0.106
        }
      ],
      "scope": {
        "filter": null,
        "n": 217,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Creator-paper v1.0 zero-shot DMS substitution setting; all five reported metrics, corrected-average aggregation, and all 50 Table A5 Spearman result rows are registered.",
        "status": "verified"
      },
      "work_id": "proteingym-paper",
      "work_version_id": "proteingym-paper-2023-12-10"
    },
    {
      "benchmark_id": "proteinlmbench",
      "benchmark_version": "paper-v2",
      "comparability_group": "proteinlmbench-paper-v2-full-official",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "proteinlmbench-eval-scope",
          "locator": {
            "note": "944-question full paper evaluation and 18 evaluated model/system labels.",
            "type": "section",
            "value": "Abstract; Sections 3.2, 4.3, 6; Appendix C.3"
          },
          "source_id": "proteinlmbench-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/benchmark_version",
            "/model_ids"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "proteinlmbench-eval-protocol",
          "locator": {
            "note": "No demonstrations/tools, single prompt, think-step-by-step instruction, temperature 0.1, 20-token cap, first-integer parser, and exact scorer.",
            "type": "repository-path",
            "value": "benchmark/benchmark_your_model.py at d8586e22ff85f6805edea0bbc23002aaccf525c4"
          },
          "source_id": "proteinlmbench-repository-resource",
          "source_type": "resource",
          "supports": [
            "/protocol"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "proteinlmbench-eval-results",
          "locator": {
            "note": "All accuracy and inference-time values; checklist confirms no repeated experiments or random seeds and no error bars.",
            "type": "table",
            "value": "p. 23, Table 3; p. 14 checklist item 3(c)"
          },
          "source_id": "proteinlmbench-paper",
          "source_type": "work",
          "supports": [
            "/protocol/seed",
            "/protocol/repeats",
            "/protocol/statistical",
            "/metrics",
            "/results"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "proteinlmbench-eval-contamination",
          "locator": {
            "note": "RAG/GPT-4 generation/validation and machine-plus-human verification; no decontamination analysis.",
            "type": "section",
            "value": "Section 4.3 and Appendix B.3 Q22"
          },
          "source_id": "proteinlmbench-paper",
          "source_type": "work",
          "supports": [
            "/protocol/contamination"
          ]
        }
      ],
      "id": "proteinlmbench-creator-full",
      "metrics": [
        {
          "aggregation": "problem-weighted over 944 questions",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Correct Rate",
          "tolerance": "exact first-integer option match",
          "unit": "percent"
        },
        {
          "aggregation": "total wall-clock time over 944 questions",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "inference-time-minutes",
          "pass_threshold": null,
          "range": [
            0,
            1000000
          ],
          "source_label": "Inference Time",
          "tolerance": null,
          "unit": "minutes"
        }
      ],
      "model_ids": [
        "proteinlmbench-baichuan2-7b",
        "proteinlmbench-chatglm3-6b",
        "proteinlmbench-falcon-7b",
        "proteinlmbench-falcon-7b-instruct",
        "proteinlmbench-gpt35-turbo",
        "proteinlmbench-gpt4-turbo",
        "proteinlmbench-internlm-chat-20b",
        "proteinlmbench-internlm2-20b",
        "proteinlmbench-internlm2-7b",
        "proteinlmbench-internlm2-chat-20b",
        "proteinlmbench-internlm2-chat-7b",
        "proteinlmbench-internlm2-protein-7b-no-ssl",
        "proteinlmbench-llama2-7b-chat",
        "proteinlmbench-mistral-7b-instruct-v02",
        "proteinlmbench-moonshot",
        "proteinlmbench-qwen15-7b",
        "proteinlmbench-yi-6b-chat",
        "toursynbio-7b"
      ],
      "protocol": {
        "contamination": {
          "notes": "GPT-4.0-turbo is also one of the evaluated models.",
          "reporting_status": "reported",
          "value": "No decontamination analysis reported; the paper says RAG generated questions and GPT-4 validated answers, while the later official repository says Mixtral-8x7B was used at each generation stage followed by expert review."
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "first-integer exact option match"
        },
        "reasoning": {
          "notes": "The output must begin with the selected option and generation is limited to 20 new tokens.",
          "reporting_status": "reported",
          "value": "Think step by step."
        },
        "repeats": {
          "notes": "The paper checklist says multiple experiments were not conducted; no error bars are reported.",
          "reporting_status": "reported",
          "value": 1
        },
        "seed": {
          "notes": "The paper checklist explicitly says random seeds were not set.",
          "reporting_status": "reported",
          "value": "not set"
        },
        "shots": {
          "notes": "The official evaluation template contains no demonstrations.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": null,
          "reporting_status": "reported",
          "value": "single-run problem-weighted accuracy and total wall-clock inference minutes; no confidence intervals or error bars"
        },
        "system_prompt_public": {
          "notes": "No separate system prompt is used; the complete user prompt template is public.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "temperature": {
          "notes": "Official runner uses sampling with temperature 0.1.",
          "reporting_status": "reported",
          "value": 0.1
        },
        "time_budget": {
          "notes": "Table 3 reports observed inference time per model, not a common enforced budget.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Official runner max_new_tokens=20.",
          "reporting_status": "reported",
          "value": "20 generated tokens per question"
        },
        "tools": {
          "browser": {
            "notes": "The official runner supplies only the question and options.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "The official runner supplies only the question and options.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No standardized container is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "The official runner supplies only the question and options.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "The official runner supplies only the question and options.",
            "reporting_status": "reported",
            "value": false
          },
          "internet": {
            "notes": "The official runner supplies only the question and options.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One prompt and one generated answer per question.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-eval-results"
          ],
          "metric_id": "accuracy",
          "model_id": "proteinlmbench-gpt4-turbo",
          "n": 944,
          "notes": "Exact API snapshot not reported.",
          "status": "verified",
          "value": 57.94
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-eval-results"
          ],
          "metric_id": "inference-time-minutes",
          "model_id": "proteinlmbench-gpt4-turbo",
          "n": 944,
          "notes": "Observed total; common hardware/service conditions not reported.",
          "status": "verified",
          "value": 15.52
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-eval-results"
          ],
          "metric_id": "accuracy",
          "model_id": "proteinlmbench-internlm2-20b",
          "n": 944,
          "notes": null,
          "status": "verified",
          "value": 57.52
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-eval-results"
          ],
          "metric_id": "inference-time-minutes",
          "model_id": "proteinlmbench-internlm2-20b",
          "n": 944,
          "notes": null,
          "status": "verified",
          "value": 47.2
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-eval-results"
          ],
          "metric_id": "accuracy",
          "model_id": "proteinlmbench-gpt35-turbo",
          "n": 944,
          "notes": "Exact API snapshot not reported.",
          "status": "verified",
          "value": 55.19
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-eval-results"
          ],
          "metric_id": "inference-time-minutes",
          "model_id": "proteinlmbench-gpt35-turbo",
          "n": 944,
          "notes": null,
          "status": "verified",
          "value": 21.03
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-eval-results"
          ],
          "metric_id": "accuracy",
          "model_id": "proteinlmbench-internlm2-7b",
          "n": 944,
          "notes": null,
          "status": "verified",
          "value": 54.98
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-eval-results"
          ],
          "metric_id": "inference-time-minutes",
          "model_id": "proteinlmbench-internlm2-7b",
          "n": 944,
          "notes": null,
          "status": "verified",
          "value": 19.23
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-eval-results"
          ],
          "metric_id": "accuracy",
          "model_id": "proteinlmbench-internlm2-chat-7b",
          "n": 944,
          "notes": null,
          "status": "verified",
          "value": 54.76
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-eval-results"
          ],
          "metric_id": "inference-time-minutes",
          "model_id": "proteinlmbench-internlm2-chat-7b",
          "n": 944,
          "notes": null,
          "status": "verified",
          "value": 35.58
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-eval-results"
          ],
          "metric_id": "accuracy",
          "model_id": "proteinlmbench-internlm2-chat-20b",
          "n": 944,
          "notes": null,
          "status": "verified",
          "value": 51.38
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-eval-results"
          ],
          "metric_id": "inference-time-minutes",
          "model_id": "proteinlmbench-internlm2-chat-20b",
          "n": 944,
          "notes": null,
          "status": "verified",
          "value": 31.11
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-eval-results"
          ],
          "metric_id": "accuracy",
          "model_id": "proteinlmbench-yi-6b-chat",
          "n": 944,
          "notes": null,
          "status": "verified",
          "value": 50.85
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-eval-results"
          ],
          "metric_id": "inference-time-minutes",
          "model_id": "proteinlmbench-yi-6b-chat",
          "n": 944,
          "notes": null,
          "status": "verified",
          "value": 59.05
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-eval-results"
          ],
          "metric_id": "accuracy",
          "model_id": "proteinlmbench-mistral-7b-instruct-v02",
          "n": 944,
          "notes": null,
          "status": "verified",
          "value": 50.11
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-eval-results"
          ],
          "metric_id": "inference-time-minutes",
          "model_id": "proteinlmbench-mistral-7b-instruct-v02",
          "n": 944,
          "notes": null,
          "status": "verified",
          "value": 13.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-eval-results"
          ],
          "metric_id": "accuracy",
          "model_id": "proteinlmbench-chatglm3-6b",
          "n": 944,
          "notes": null,
          "status": "verified",
          "value": 48.94
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-eval-results"
          ],
          "metric_id": "inference-time-minutes",
          "model_id": "proteinlmbench-chatglm3-6b",
          "n": 944,
          "notes": null,
          "status": "verified",
          "value": 8.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-eval-results"
          ],
          "metric_id": "accuracy",
          "model_id": "proteinlmbench-baichuan2-7b",
          "n": 944,
          "notes": null,
          "status": "verified",
          "value": 44.49
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-eval-results"
          ],
          "metric_id": "inference-time-minutes",
          "model_id": "proteinlmbench-baichuan2-7b",
          "n": 944,
          "notes": null,
          "status": "verified",
          "value": 16.37
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-eval-results"
          ],
          "metric_id": "accuracy",
          "model_id": "proteinlmbench-internlm-chat-20b",
          "n": 944,
          "notes": null,
          "status": "verified",
          "value": 40.54
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-eval-results"
          ],
          "metric_id": "inference-time-minutes",
          "model_id": "proteinlmbench-internlm-chat-20b",
          "n": 944,
          "notes": null,
          "status": "verified",
          "value": 66.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-eval-results"
          ],
          "metric_id": "accuracy",
          "model_id": "proteinlmbench-llama2-7b-chat",
          "n": 944,
          "notes": null,
          "status": "verified",
          "value": 39.64
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-eval-results"
          ],
          "metric_id": "inference-time-minutes",
          "model_id": "proteinlmbench-llama2-7b-chat",
          "n": 944,
          "notes": null,
          "status": "verified",
          "value": 64.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-eval-results"
          ],
          "metric_id": "accuracy",
          "model_id": "proteinlmbench-moonshot",
          "n": 944,
          "notes": "Exact provider model/version not reported.",
          "status": "verified",
          "value": 38.26
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-eval-results"
          ],
          "metric_id": "inference-time-minutes",
          "model_id": "proteinlmbench-moonshot",
          "n": 944,
          "notes": null,
          "status": "verified",
          "value": 16.25
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-eval-results"
          ],
          "metric_id": "accuracy",
          "model_id": "proteinlmbench-qwen15-7b",
          "n": 944,
          "notes": null,
          "status": "verified",
          "value": 21.73
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-eval-results"
          ],
          "metric_id": "inference-time-minutes",
          "model_id": "proteinlmbench-qwen15-7b",
          "n": 944,
          "notes": null,
          "status": "verified",
          "value": 13.0
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-eval-results"
          ],
          "metric_id": "accuracy",
          "model_id": "proteinlmbench-falcon-7b-instruct",
          "n": 944,
          "notes": null,
          "status": "verified",
          "value": 20.55
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-eval-results"
          ],
          "metric_id": "inference-time-minutes",
          "model_id": "proteinlmbench-falcon-7b-instruct",
          "n": 944,
          "notes": null,
          "status": "verified",
          "value": 25.42
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-eval-results"
          ],
          "metric_id": "accuracy",
          "model_id": "proteinlmbench-falcon-7b",
          "n": 944,
          "notes": null,
          "status": "verified",
          "value": 19.17
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-eval-results"
          ],
          "metric_id": "inference-time-minutes",
          "model_id": "proteinlmbench-falcon-7b",
          "n": 944,
          "notes": null,
          "status": "verified",
          "value": 15.55
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-eval-results"
          ],
          "metric_id": "accuracy",
          "model_id": "toursynbio-7b",
          "n": 944,
          "notes": "InternLM2-Protein-7B with SSL then SFT.",
          "status": "verified",
          "value": 62.18
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-eval-results"
          ],
          "metric_id": "inference-time-minutes",
          "model_id": "toursynbio-7b",
          "n": 944,
          "notes": null,
          "status": "verified",
          "value": 22.34
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-eval-results"
          ],
          "metric_id": "accuracy",
          "model_id": "proteinlmbench-internlm2-protein-7b-no-ssl",
          "n": 944,
          "notes": "SFT only; no ProteinLMDataset self-supervised phase.",
          "status": "verified",
          "value": 58.26
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "proteinlmbench-eval-results"
          ],
          "metric_id": "inference-time-minutes",
          "model_id": "proteinlmbench-internlm2-protein-7b-no-ssl",
          "n": 944,
          "notes": null,
          "status": "verified",
          "value": 21.36
        }
      ],
      "scope": {
        "filter": "All 944 paper-defined ProteinLMBench questions; the paper does not pin an exact Hugging Face commit.",
        "n": 944,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "All 18 model labels, 36 Table 3 values, paper-defined scope, public runner settings, and missing version/seed/error-bar details are explicitly recorded.",
        "status": "verified"
      },
      "work_id": "proteinlmbench-paper",
      "work_version_id": "proteinlmbench-paper-2024-06-08"
    },
    {
      "benchmark_id": "scib",
      "benchmark_version": "paper-2021",
      "comparability_group": "scib-paper-2021-native",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "scib-creator-full-protocol-evidence",
          "locator": {
            "note": "Defines tasks, methods, preprocessing, fourteen raw metrics, rescaling, category means, and 60/40 overall score.",
            "type": "section",
            "value": "Results: scIB; Figure 1; Table 1; Methods: Evaluation metrics and Metric aggregation"
          },
          "source_id": "scib-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/protocol",
            "/metrics"
          ]
        }
      ],
      "id": "scib-creator-full",
      "metrics": [
        {
          "aggregation": "batch removal within labels",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "kbet",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "kBET",
          "tolerance": null,
          "unit": "normalized score"
        },
        {
          "aggregation": "batch removal within labels",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "graph-connectivity",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Graph connectivity",
          "tolerance": null,
          "unit": "normalized score"
        },
        {
          "aggregation": "batch removal within labels",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "batch-asw",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Batch ASW",
          "tolerance": null,
          "unit": "normalized score"
        },
        {
          "aggregation": "label-independent batch removal",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "graph-ilisi",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Graph iLISI",
          "tolerance": null,
          "unit": "normalized score"
        },
        {
          "aggregation": "label-independent batch removal",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "pcr-comparison",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "PCA regression",
          "tolerance": null,
          "unit": "normalized score"
        },
        {
          "aggregation": "label conservation",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "graph-clisi",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Graph cLISI",
          "tolerance": null,
          "unit": "normalized score"
        },
        {
          "aggregation": "cluster-label agreement",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "adjusted-rand-index",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "ARI",
          "tolerance": null,
          "unit": "normalized score"
        },
        {
          "aggregation": "cluster-label agreement",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "normalized-mutual-information",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "NMI",
          "tolerance": null,
          "unit": "normalized score"
        },
        {
          "aggregation": "label conservation",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "cell-type-asw",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Cell-type ASW",
          "tolerance": null,
          "unit": "normalized score"
        },
        {
          "aggregation": "rare labels",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "isolated-label-f1",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Isolated-label F1",
          "tolerance": null,
          "unit": "normalized score"
        },
        {
          "aggregation": "rare labels",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "isolated-label-asw",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Isolated-label ASW",
          "tolerance": null,
          "unit": "normalized score"
        },
        {
          "aggregation": "label-free biological conservation",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "cell-cycle-conservation",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Cell-cycle conservation",
          "tolerance": null,
          "unit": "normalized score"
        },
        {
          "aggregation": "label-free biological conservation",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "hvg-conservation",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "HVG conservation",
          "tolerance": null,
          "unit": "normalized score"
        },
        {
          "aggregation": "label-free biological conservation",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "trajectory-conservation",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Trajectory conservation",
          "tolerance": null,
          "unit": "normalized score"
        },
        {
          "aggregation": "mean of applicable batch metrics",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "batch-removal-score",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Batch-removal score",
          "tolerance": null,
          "unit": "normalized score"
        },
        {
          "aggregation": "mean of applicable biological metrics",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "bio-conservation-score",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Bio-conservation score",
          "tolerance": null,
          "unit": "normalized score"
        },
        {
          "aggregation": "0.6 bio-conservation plus 0.4 batch-removal",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "overall-integration-score",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Overall score",
          "tolerance": null,
          "unit": "normalized score"
        }
      ],
      "model_ids": [],
      "protocol": {
        "contamination": {
          "notes": "Preprocessing variations are explicit evaluation factors.",
          "reporting_status": "reported",
          "value": "Task-specific preprocessing and pre-annotation establish biological labels; simulations provide known ground truth."
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "deterministic single-cell integration metric suite"
        },
        "reasoning": {
          "notes": "Data-integration methods; no prompting.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "repeats": {
          "notes": "No common repeat count applies across methods and tasks.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "No single common seed is reported across all external tools.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Data-integration methods; no prompting.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "statistical": {
          "notes": "Category scores average applicable metrics so output types with unavailable metrics remain comparable.",
          "reporting_status": "reported",
          "value": "Raw metrics are rescaled to higher-is-better; overall = 0.6 bio-conservation + 0.4 batch-removal."
        },
        "system_prompt_public": {
          "notes": "Data-integration methods; no prompting.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "temperature": {
          "notes": "Data-integration methods; no prompting.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "time_budget": {
          "notes": "Scalability is measured rather than enforced as a shared timeout.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Data-integration methods; no prompting.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Data-integration methods; no prompting.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "code_execution": {
            "notes": "Integration and metric code is the evaluation procedure.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "container": {
            "notes": "Reproducibility workflow isolates method dependencies.",
            "reporting_status": "reported",
            "value": "task and method-specific environments in the Snakemake pipeline"
          },
          "databases": {
            "notes": "Data-integration methods; no prompting.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "external_tools": {
            "notes": "Outputs are treated as separate integration runs.",
            "reporting_status": "reported",
            "value": "16 integration methods with four preprocessing combinations"
          },
          "internet": {
            "notes": "Data-integration methods; no prompting.",
            "reporting_status": "not_applicable",
            "value": null
          }
        },
        "turns": {
          "notes": "Data-integration methods; no prompting.",
          "reporting_status": "not_applicable",
          "value": null
        }
      },
      "results": [],
      "scope": {
        "filter": "Two simulation, five scRNA-seq, and six scATAC-seq integration tasks.",
        "n": 13,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Full 13-task scope, 16-method/four-preprocessing design, fourteen raw metrics, and weighted aggregation are verified.",
        "status": "verified"
      },
      "work_id": "scib-paper",
      "work_version_id": "scib-paper-2021-12-23"
    },
    {
      "benchmark_id": "scigym-small",
      "benchmark_version": "2025 release",
      "comparability_group": "scigym-2025-small-react-initial-concentration-20step-3repeat",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "scigym-run-evidence-main-protocol",
          "locator": {
            "note": "Defines the ReAct loop, three action types, public prompt, initial-concentration experiment, 20 action iterations, three debugging iterations, six exact model versions, full 137-system small scope, and NTS/RMS/STE.",
            "type": "section",
            "value": "NeurIPS paper §§3.2–3.3 and 5; Appendix A"
          },
          "source_id": "scigym-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/scope",
            "/model_ids",
            "/protocol/shots",
            "/protocol/turns",
            "/protocol/system_prompt_public",
            "/protocol/reasoning",
            "/protocol/tools",
            "/protocol/time_budget",
            "/protocol/grader",
            "/protocol/statistical",
            "/protocol/contamination",
            "/metrics",
            "/comparability_group"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "scigym-run-evidence-public-episodes",
          "locator": {
            "note": "2,466 rows; 137 unique systems, six exact model strings, and exactly three rows for every model-system pair.",
            "type": "dataset-card",
            "value": "data/small-00000-of-00001.parquet at commit 7d472c12855d46702c4915892578290355894c1a"
          },
          "source_id": "scigym-small-results-resource",
          "source_type": "resource",
          "supports": [
            "/model_ids",
            "/protocol/repeats",
            "/protocol/statistical"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "scigym-run-evidence-code",
          "locator": {
            "note": "Confirms exact API strings, maximum 8,192 response tokens, no browser/network/database tool, public Python/SBML tool loop, and deterministic evaluator.",
            "type": "repository-path",
            "value": "README.md, scigym/llm.py, scigym/controller.py, scigym/system_prompts/, and scigym/evaluator.py at commit d290bb04bf54aad1c473c4701e0d0d88013c4f91"
          },
          "source_id": "scigym-small-evaluation-code-resource",
          "source_type": "resource",
          "supports": [
            "/model_ids",
            "/protocol/system_prompt_public",
            "/protocol/tools",
            "/protocol/token_budget",
            "/protocol/grader",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-21",
          "id": "scigym-run-evidence-main-results",
          "locator": {
            "note": "Prints STE and RMS precision/recall/F1 with and without modifiers for all six models; no confidence bounds are printed.",
            "type": "table",
            "value": "NeurIPS paper Table 1"
          },
          "source_id": "scigym-paper",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "scigym-small-creator-paper",
      "metrics": [
        {
          "aggregation": "pairwise species-interaction F1 with duplicate relationships counted once, then averaged across systems/episodes",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "network-topology-f1",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Network Topology Score (NTS) F1",
          "tolerance": "relationship type includes reactant-product, reactant-modifier, and modifier-product",
          "unit": "score"
        },
        {
          "aggregation": "SMAPE averaged across species and benchmark systems/episodes",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "simulation-trajectory-error",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Simulation Trajectory Error (STE)",
          "tolerance": "evaluated under original and perturbed initial conditions",
          "unit": "SMAPE error"
        },
        {
          "aggregation": "exact reactant/product/modifier reaction matching averaged across systems/episodes",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "reaction-matching-with-modifiers-precision",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "RMS with modifiers — Precision",
          "tolerance": "reactants, products, and modifiers must match",
          "unit": "score"
        },
        {
          "aggregation": "exact reactant/product/modifier reaction matching averaged across systems/episodes",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "reaction-matching-with-modifiers-recall",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "RMS with modifiers — Recall",
          "tolerance": "reactants, products, and modifiers must match",
          "unit": "score"
        },
        {
          "aggregation": "harmonic mean of with-modifier reaction precision and recall averaged across systems/episodes",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "reaction-matching-with-modifiers-f1",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "RMS with modifiers — F1",
          "tolerance": "reactants, products, and modifiers must match",
          "unit": "score"
        },
        {
          "aggregation": "exact reactant/product reaction matching averaged across systems/episodes",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "reaction-matching-without-modifiers-precision",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "RMS without modifiers — Precision",
          "tolerance": "modifiers are ignored",
          "unit": "score"
        },
        {
          "aggregation": "exact reactant/product reaction matching averaged across systems/episodes",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "reaction-matching-without-modifiers-recall",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "RMS without modifiers — Recall",
          "tolerance": "modifiers are ignored",
          "unit": "score"
        },
        {
          "aggregation": "harmonic mean of without-modifier reaction precision and recall averaged across systems/episodes",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "reaction-matching-without-modifiers-f1",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "RMS without modifiers — F1",
          "tolerance": "modifiers are ignored",
          "unit": "score"
        }
      ],
      "model_ids": [
        "scigym-claude-3-5-haiku-20241022",
        "scigym-claude-3-7-sonnet-20250219",
        "scigym-gemini-2-5-flash-preview-04-17",
        "scigym-gemini-2-5-pro-preview-03-25",
        "scigym-gpt-4-1-2025-04-14",
        "scigym-gpt-4-1-mini-2025-04-14"
      ],
      "protocol": {
        "contamination": {
          "notes": "This is a task-leakage mitigation, not a formal pretraining-contamination analysis.",
          "reporting_status": "reported",
          "value": "Reference models are de-identified by stripping metadata, shuffling components, and replacing component IDs; species names are retained. No model-training decontamination test is reported."
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "deterministic SBML structural and simulation evaluator"
        },
        "reasoning": {
          "notes": "The agent must state thoughts before actions and is not shown the evaluation metrics.",
          "reporting_status": "reported",
          "value": "ReAct-style Thoughts–Actions–Observations agent; no provider reasoning-effort control is reported."
        },
        "repeats": {
          "notes": "The immutable evaluation Parquet contains exactly three rows for every one of the 137 system × 6 model pairs (2,466 total episodes).",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "No model-sampling seed is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "No benchmark-episode shot count is reported. The public prompt contains SBML and tool-usage examples, not a solved SCIGYM instance.",
          "reporting_status": "not_reported",
          "value": null
        },
        "statistical": {
          "notes": "NTS and RMS use precision/recall/F1 matching; STE averages species-level SMAPE and evaluates original and perturbed initial conditions.",
          "reporting_status": "reported",
          "value": "Table 1 reports arithmetic means across the small benchmark instances. The public artifact supplies three episodes for every model-system pair; Table 1 does not print confidence intervals."
        },
        "system_prompt_public": {
          "notes": "System prompt, experiment manual, customized function documentation, and output formats are published in the paper appendix and creator repository.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "The main-run temperature is absent from the paper and released trajectories. Example/default repository configs disagree and are not treated as the experimental value.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "An agent may submit earlier; after a malformed or non-simulatable model, up to three additional debugging turns are allowed.",
          "reporting_status": "reported",
          "value": "maximum 20 action iterations plus up to 3 invalid-submission debugging iterations"
        },
        "token_budget": {
          "notes": "The public evaluation-era LLM wrapper initializes max_length=8192.",
          "reporting_status": "reported",
          "value": "maximum 8,192 output tokens per model response"
        },
        "tools": {
          "browser": {
            "notes": "No browser is exposed.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "Stateful Python shell for analysis and SBML construction.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "The sources describe a Python execution environment but do not report container isolation.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "The BioModels source is curated before evaluation; the agent has no live database tool.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Initial-concentration perturbations are simulated from the hidden reference model and returned as pandas time-series data.",
            "reporting_status": "reported",
            "value": [
              "Tellurium",
              "libRoadRunner",
              "libSBML",
              "pandas",
              "numpy",
              "SCIGYM experiment API"
            ]
          },
          "internet": {
            "notes": "No network access is exposed during an episode.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "ReAct-style Thoughts–Actions–Observations loop with code, experiment, and submit actions.",
          "reporting_status": "reported",
          "value": "multi-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "simulation-trajectory-error",
          "model_id": "scigym-gemini-2-5-flash-preview-04-17",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.4181
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "reaction-matching-with-modifiers-precision",
          "model_id": "scigym-gemini-2-5-flash-preview-04-17",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.1527
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "reaction-matching-with-modifiers-recall",
          "model_id": "scigym-gemini-2-5-flash-preview-04-17",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.1071
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "reaction-matching-with-modifiers-f1",
          "model_id": "scigym-gemini-2-5-flash-preview-04-17",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.1217
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "reaction-matching-without-modifiers-precision",
          "model_id": "scigym-gemini-2-5-flash-preview-04-17",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.2399
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "reaction-matching-without-modifiers-recall",
          "model_id": "scigym-gemini-2-5-flash-preview-04-17",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.1839
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "reaction-matching-without-modifiers-f1",
          "model_id": "scigym-gemini-2-5-flash-preview-04-17",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.2005
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "simulation-trajectory-error",
          "model_id": "scigym-gpt-4-1-mini-2025-04-14",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.6007
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "reaction-matching-with-modifiers-precision",
          "model_id": "scigym-gpt-4-1-mini-2025-04-14",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.1516
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "reaction-matching-with-modifiers-recall",
          "model_id": "scigym-gpt-4-1-mini-2025-04-14",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.1253
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "reaction-matching-with-modifiers-f1",
          "model_id": "scigym-gpt-4-1-mini-2025-04-14",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.132
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "reaction-matching-without-modifiers-precision",
          "model_id": "scigym-gpt-4-1-mini-2025-04-14",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.253
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "reaction-matching-without-modifiers-recall",
          "model_id": "scigym-gpt-4-1-mini-2025-04-14",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.2313
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "reaction-matching-without-modifiers-f1",
          "model_id": "scigym-gpt-4-1-mini-2025-04-14",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.2322
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "simulation-trajectory-error",
          "model_id": "scigym-claude-3-5-haiku-20241022",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.6281
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "reaction-matching-with-modifiers-precision",
          "model_id": "scigym-claude-3-5-haiku-20241022",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.0858
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "reaction-matching-with-modifiers-recall",
          "model_id": "scigym-claude-3-5-haiku-20241022",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.0421
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "reaction-matching-with-modifiers-f1",
          "model_id": "scigym-claude-3-5-haiku-20241022",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.053
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "reaction-matching-without-modifiers-precision",
          "model_id": "scigym-claude-3-5-haiku-20241022",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.1454
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "reaction-matching-without-modifiers-recall",
          "model_id": "scigym-claude-3-5-haiku-20241022",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.0805
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "reaction-matching-without-modifiers-f1",
          "model_id": "scigym-claude-3-5-haiku-20241022",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.0987
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "simulation-trajectory-error",
          "model_id": "scigym-gemini-2-5-pro-preview-03-25",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.3212
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "reaction-matching-with-modifiers-precision",
          "model_id": "scigym-gemini-2-5-pro-preview-03-25",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.2138
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "reaction-matching-with-modifiers-recall",
          "model_id": "scigym-gemini-2-5-pro-preview-03-25",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.1664
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "reaction-matching-with-modifiers-f1",
          "model_id": "scigym-gemini-2-5-pro-preview-03-25",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.1817
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "reaction-matching-without-modifiers-precision",
          "model_id": "scigym-gemini-2-5-pro-preview-03-25",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.3781
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "reaction-matching-without-modifiers-recall",
          "model_id": "scigym-gemini-2-5-pro-preview-03-25",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.3219
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "reaction-matching-without-modifiers-f1",
          "model_id": "scigym-gemini-2-5-pro-preview-03-25",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.3383
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "simulation-trajectory-error",
          "model_id": "scigym-gpt-4-1-2025-04-14",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.4611
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "reaction-matching-with-modifiers-precision",
          "model_id": "scigym-gpt-4-1-2025-04-14",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.2067
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "reaction-matching-with-modifiers-recall",
          "model_id": "scigym-gpt-4-1-2025-04-14",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.1597
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "reaction-matching-with-modifiers-f1",
          "model_id": "scigym-gpt-4-1-2025-04-14",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.174
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "reaction-matching-without-modifiers-precision",
          "model_id": "scigym-gpt-4-1-2025-04-14",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.3517
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "reaction-matching-without-modifiers-recall",
          "model_id": "scigym-gpt-4-1-2025-04-14",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.2888
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "reaction-matching-without-modifiers-f1",
          "model_id": "scigym-gpt-4-1-2025-04-14",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.3038
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "simulation-trajectory-error",
          "model_id": "scigym-claude-3-7-sonnet-20250219",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.3615
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "reaction-matching-with-modifiers-precision",
          "model_id": "scigym-claude-3-7-sonnet-20250219",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.178
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "reaction-matching-with-modifiers-recall",
          "model_id": "scigym-claude-3-7-sonnet-20250219",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.1698
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "reaction-matching-with-modifiers-f1",
          "model_id": "scigym-claude-3-7-sonnet-20250219",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.1688
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "reaction-matching-without-modifiers-precision",
          "model_id": "scigym-claude-3-7-sonnet-20250219",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.316
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "reaction-matching-without-modifiers-recall",
          "model_id": "scigym-claude-3-7-sonnet-20250219",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.317
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "scigym-run-evidence-main-results"
          ],
          "metric_id": "reaction-matching-without-modifiers-f1",
          "model_id": "scigym-claude-3-7-sonnet-20250219",
          "n": 137,
          "notes": "Table 1; three public episodes per system.",
          "status": "verified",
          "value": 0.3047
        }
      ],
      "scope": {
        "filter": "All 137 systems in the official small split, each having fewer than ten reactions.",
        "n": 137,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Full small-track scope, six exact model versions, multi-turn environment, 20+3 turn budget, deterministic metrics, three public episodes per pair, and all 42 Table 1 values were verified.",
        "status": "verified"
      },
      "work_id": "scigym-paper",
      "work_version_id": "scigym-paper-2025-11-30"
    },
    {
      "benchmark_id": "scigym-small",
      "benchmark_version": "2025 release",
      "comparability_group": "scigym-2025-small-zero-shot-no-tools",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-21",
          "id": "scigym-zero-shot-evidence",
          "locator": {
            "note": "Compares each of the six agents against a zero-shot/direct-prompt baseline across small systems, with experiment and tools removed and three debugging rounds; exact aggregates are not tabulated.",
            "type": "figure",
            "value": "NeurIPS paper Figure 5 and §5.1"
          },
          "source_id": "scigym-paper",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/model_ids",
            "/scope",
            "/protocol",
            "/metrics",
            "/results",
            "/comparability_group"
          ]
        }
      ],
      "id": "scigym-small-zero-shot",
      "metrics": [
        {
          "aggregation": "per-system structural F1",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "network-topology-f1",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Network Topology Score (NTS) F1",
          "tolerance": "duplicate relationships counted once",
          "unit": "score"
        },
        {
          "aggregation": "species-level SMAPE averaged within system",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "simulation-trajectory-error",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Simulation Trajectory Error (STE)",
          "tolerance": "original and perturbed initial conditions",
          "unit": "SMAPE error"
        },
        {
          "aggregation": "exact-reaction F1",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "reaction-matching-with-modifiers-f1",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "RMS with modifiers — F1",
          "tolerance": "reactants products and modifiers must match",
          "unit": "score"
        },
        {
          "aggregation": "exact-reaction F1",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "reaction-matching-without-modifiers-f1",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "RMS without modifiers — F1",
          "tolerance": "modifiers ignored",
          "unit": "score"
        }
      ],
      "model_ids": [
        "scigym-claude-3-5-haiku-20241022",
        "scigym-claude-3-7-sonnet-20250219",
        "scigym-gemini-2-5-flash-preview-04-17",
        "scigym-gemini-2-5-pro-preview-03-25",
        "scigym-gpt-4-1-2025-04-14",
        "scigym-gpt-4-1-mini-2025-04-14"
      ],
      "protocol": {
        "contamination": {
          "notes": null,
          "reporting_status": "reported",
          "value": "The same de-identified small systems are used; no model-training decontamination test is reported."
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "deterministic SBML structural and simulation evaluator"
        },
        "reasoning": {
          "notes": "The prompt retains comparable task framing but excludes the dry-lab tool loop.",
          "reporting_status": "reported",
          "value": "direct prompting without experimental feedback"
        },
        "repeats": {
          "notes": "The number of direct-baseline samples per model-system pair is not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "No seed is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The creator paper explicitly calls this the zero-shot/direct-prompt baseline.",
          "reporting_status": "reported",
          "value": "zero-shot"
        },
        "statistical": {
          "notes": "No numeric result rows are registered from unlabeled plot markers.",
          "reporting_status": "reported",
          "value": "Figure 5 plots per-system agent and zero-shot scores; exact aggregate baseline values and confidence intervals are not tabulated."
        },
        "system_prompt_public": {
          "notes": "A similar ReAct-format prompt is used with experiment and analysis tools removed.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "The baseline temperature is not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Three additional debugging rounds are reported.",
          "reporting_status": "reported",
          "value": "one direct submission plus up to 3 invalid-submission debugging iterations"
        },
        "token_budget": {
          "notes": "Evaluation-era wrapper default.",
          "reporting_status": "reported",
          "value": "maximum 8,192 output tokens per model response"
        },
        "tools": {
          "browser": {
            "notes": "No browser.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "The model has no code-analysis tool in this baseline; submitted SBML is evaluated after generation.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "No interactive execution environment is exposed to the model.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "No database.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "Experiment and analysis tools are removed.",
            "reporting_status": "reported",
            "value": "none"
          },
          "internet": {
            "notes": "No internet.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "This is not the iterative experiment loop.",
          "reporting_status": "reported",
          "value": "one direct submission, with up to three debugging responses after invalid submissions"
        }
      },
      "results": [],
      "scope": {
        "filter": "All 137 small systems in the direct-prompt baseline shown in Figure 5.",
        "n": 137,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "The official full-small direct baseline and its no-tool/three-debug protocol are registered, but unlabeled Figure 5 values are intentionally not digitized.",
        "status": "verified"
      },
      "work_id": "scigym-paper",
      "work_version_id": "scigym-paper-2025-11-30"
    },
    {
      "benchmark_id": "single-cell-omics-arena-soar",
      "benchmark_version": "initial-release",
      "comparability_group": "soar-e5d2b3e-rna-zero-shot",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-08-13",
          "id": "soar-e5d2b3e-rna-zero-shot-protocol-evidence",
          "locator": {
            "note": "Scope, exact models, zero-shot prompt, model call, budgets, tools, and automatic metric implementation.",
            "type": "repository-path",
            "value": "soar_benchmark/datasets/soar_rna.json; soar_benchmark/configs/cell_type_annotation/experiment_soar_rna.py; soar_benchmark/task.py; soar_benchmark/pipeline.py; soar_benchmark/prompt_templates/factory.py; analysis/cell_type_annotation/eval_multiple.py at commit e5d2b3e2619cb56fece5fba78fae989a67fd0c13"
          },
          "source_id": "single-cell-omics-arena-soar-official-repository-resource",
          "source_type": "resource",
          "supports": [
            "/benchmark_version",
            "/scope",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-08-13",
          "id": "soar-e5d2b3e-rna-zero-shot-results-evidence",
          "locator": {
            "note": "Exact GPT-4o and GPT-4o mini values for the seven printed metrics.",
            "type": "repository-path",
            "value": "readme.md, SOAR-RNA Benchmark, Zero-shot Cell Type Annotation table at commit e5d2b3e2619cb56fece5fba78fae989a67fd0c13"
          },
          "source_id": "single-cell-omics-arena-soar-official-repository-resource",
          "source_type": "resource",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "soar-e5d2b3e-rna-zero-shot",
      "metrics": [
        {
          "aggregation": "all predictions and normalized references",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "rouge-1",
          "pass_threshold": null,
          "range": null,
          "source_label": "R-1",
          "tolerance": null,
          "unit": "source-reported score"
        },
        {
          "aggregation": "all predictions and normalized references",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "rouge-2",
          "pass_threshold": null,
          "range": null,
          "source_label": "R-2",
          "tolerance": null,
          "unit": "source-reported score"
        },
        {
          "aggregation": "all predictions and normalized references",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "rouge-l",
          "pass_threshold": null,
          "range": null,
          "source_label": "R-L",
          "tolerance": null,
          "unit": "source-reported score"
        },
        {
          "aggregation": "all predictions and normalized references",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "meteor",
          "pass_threshold": null,
          "range": null,
          "source_label": "MET.",
          "tolerance": null,
          "unit": "source-reported score"
        },
        {
          "aggregation": "all predictions and normalized references; maximum order two",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "bleu-1",
          "pass_threshold": null,
          "range": null,
          "source_label": "B-1",
          "tolerance": null,
          "unit": "source-reported score"
        },
        {
          "aggregation": "all predictions and normalized references; maximum order two",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "bleu-2",
          "pass_threshold": null,
          "range": null,
          "source_label": "B-2",
          "tolerance": null,
          "unit": "source-reported score"
        },
        {
          "aggregation": "geometric average through order two over all predictions and normalized references",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "bleu",
          "pass_threshold": null,
          "range": null,
          "source_label": "BLEU",
          "tolerance": null,
          "unit": "source-reported score"
        }
      ],
      "model_ids": [
        "openai-gpt-4o-2024-05-13",
        "openai-gpt-4o-mini-2024-07-18"
      ],
      "protocol": {
        "contamination": {
          "notes": "No contamination or decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": null,
          "model": null,
          "reporting_status": "reported",
          "type": "automatic evaluate-library text-overlap metrics"
        },
        "reasoning": {
          "notes": "The zero-shot prompt requests a direct cell-type answer without a chain-of-thought stage.",
          "reporting_status": "reported",
          "value": "none"
        },
        "repeats": {
          "notes": "The published code traverses the artifact once, but the source does not label a formal evaluation repeat count.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "TaskConfig defines a local framework seed, but the OpenAI request passes no API seed.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The official configuration selects the zero_shot prompt template.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "No confidence interval or repeated-run aggregation is reported.",
          "reporting_status": "reported",
          "value": "metrics computed over all predictions and normalized references"
        },
        "system_prompt_public": {
          "notes": "The system and user prompt templates are published in the pinned repository.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "The OpenAI chat completion call explicitly sets temperature=0.",
          "reporting_status": "reported",
          "value": 0
        },
        "time_budget": {
          "notes": "The pinned repository does not report a per-example time budget.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "GenerationConfig(max_new_tokens=1024) is forwarded as max_tokens.",
          "reporting_status": "reported",
          "value": "1024 maximum output tokens"
        },
        "tools": {
          "browser": {
            "notes": "The published ChatGPT pipeline passes only messages to the model and declares no tool interface.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "The published ChatGPT pipeline passes only messages to the model and declares no tool interface.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "The published ChatGPT pipeline passes only messages to the model and declares no tool interface.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "The published ChatGPT pipeline passes only messages to the model and declares no tool interface.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "No external model tools are declared by the published pipeline.",
            "reporting_status": "reported",
            "value": []
          },
          "internet": {
            "notes": "The published ChatGPT pipeline passes only messages to the model and declares no tool interface.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "One OpenAI chat completion is retained per SOAR-RNA entry.",
          "reporting_status": "reported",
          "value": "single-turn"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "soar-e5d2b3e-rna-zero-shot-results-evidence"
          ],
          "metric_id": "rouge-1",
          "model_id": "openai-gpt-4o-mini-2024-07-18",
          "n": 1191,
          "notes": "SOAR-RNA zero-shot README table.",
          "status": "verified",
          "value": 52.63
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "soar-e5d2b3e-rna-zero-shot-results-evidence"
          ],
          "metric_id": "rouge-2",
          "model_id": "openai-gpt-4o-mini-2024-07-18",
          "n": 1191,
          "notes": "SOAR-RNA zero-shot README table.",
          "status": "verified",
          "value": 27.45
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "soar-e5d2b3e-rna-zero-shot-results-evidence"
          ],
          "metric_id": "rouge-l",
          "model_id": "openai-gpt-4o-mini-2024-07-18",
          "n": 1191,
          "notes": "SOAR-RNA zero-shot README table.",
          "status": "verified",
          "value": 52.26
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "soar-e5d2b3e-rna-zero-shot-results-evidence"
          ],
          "metric_id": "meteor",
          "model_id": "openai-gpt-4o-mini-2024-07-18",
          "n": 1191,
          "notes": "SOAR-RNA zero-shot README table.",
          "status": "verified",
          "value": 41.08
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "soar-e5d2b3e-rna-zero-shot-results-evidence"
          ],
          "metric_id": "bleu-1",
          "model_id": "openai-gpt-4o-mini-2024-07-18",
          "n": 1191,
          "notes": "SOAR-RNA zero-shot README table.",
          "status": "verified",
          "value": 45.74
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "soar-e5d2b3e-rna-zero-shot-results-evidence"
          ],
          "metric_id": "bleu-2",
          "model_id": "openai-gpt-4o-mini-2024-07-18",
          "n": 1191,
          "notes": "SOAR-RNA zero-shot README table.",
          "status": "verified",
          "value": 23.29
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "soar-e5d2b3e-rna-zero-shot-results-evidence"
          ],
          "metric_id": "bleu",
          "model_id": "openai-gpt-4o-mini-2024-07-18",
          "n": 1191,
          "notes": "SOAR-RNA zero-shot README table.",
          "status": "verified",
          "value": 32.64
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "soar-e5d2b3e-rna-zero-shot-results-evidence"
          ],
          "metric_id": "rouge-1",
          "model_id": "openai-gpt-4o-2024-05-13",
          "n": 1191,
          "notes": "SOAR-RNA zero-shot README table.",
          "status": "verified",
          "value": 58.45
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "soar-e5d2b3e-rna-zero-shot-results-evidence"
          ],
          "metric_id": "rouge-2",
          "model_id": "openai-gpt-4o-2024-05-13",
          "n": 1191,
          "notes": "SOAR-RNA zero-shot README table.",
          "status": "verified",
          "value": 32.07
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "soar-e5d2b3e-rna-zero-shot-results-evidence"
          ],
          "metric_id": "rouge-l",
          "model_id": "openai-gpt-4o-2024-05-13",
          "n": 1191,
          "notes": "SOAR-RNA zero-shot README table.",
          "status": "verified",
          "value": 58.12
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "soar-e5d2b3e-rna-zero-shot-results-evidence"
          ],
          "metric_id": "meteor",
          "model_id": "openai-gpt-4o-2024-05-13",
          "n": 1191,
          "notes": "SOAR-RNA zero-shot README table.",
          "status": "verified",
          "value": 45.39
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "soar-e5d2b3e-rna-zero-shot-results-evidence"
          ],
          "metric_id": "bleu-1",
          "model_id": "openai-gpt-4o-2024-05-13",
          "n": 1191,
          "notes": "SOAR-RNA zero-shot README table.",
          "status": "verified",
          "value": 62.85
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "soar-e5d2b3e-rna-zero-shot-results-evidence"
          ],
          "metric_id": "bleu-2",
          "model_id": "openai-gpt-4o-2024-05-13",
          "n": 1191,
          "notes": "SOAR-RNA zero-shot README table.",
          "status": "verified",
          "value": 42.68
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "soar-e5d2b3e-rna-zero-shot-results-evidence"
          ],
          "metric_id": "bleu",
          "model_id": "openai-gpt-4o-2024-05-13",
          "n": 1191,
          "notes": "SOAR-RNA zero-shot README table.",
          "status": "verified",
          "value": 51.79
        }
      ],
      "scope": {
        "filter": "All records in the pinned SOAR-RNA artifact, iterated in repository order with shuffle disabled.",
        "n": 1191,
        "reporting_status": "reported",
        "selection": "formal-subset",
        "subset_id": "soar-rna",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-08-13",
        "notes": "Formal-subset protocol and all 14 table values passed independent high-confidence verification.",
        "status": "verified"
      },
      "work_id": "single-cell-omics-arena-soar-repository-result-snapshot",
      "work_version_id": "single-cell-omics-arena-soar-repository-result-snapshot-e5d2b3e"
    },
    {
      "benchmark_id": "single-cell-omics-arena-soar",
      "benchmark_version": "initial-release",
      "comparability_group": "soar-e5d2b3e-rna-zero-shot-cot",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-08-13",
          "id": "soar-e5d2b3e-rna-zero-shot-cot-protocol-evidence",
          "locator": {
            "note": "Scope, exact models, two-call CoT prompt, model calls, budgets, tools, and automatic metric implementation.",
            "type": "repository-path",
            "value": "soar_benchmark/datasets/soar_rna.json; soar_benchmark/configs/cell_type_annotation/experiment_soar_rna.py; soar_benchmark/task.py; soar_benchmark/pipeline.py; soar_benchmark/prompt_templates/factory.py; analysis/cell_type_annotation/eval_multiple.py at commit e5d2b3e2619cb56fece5fba78fae989a67fd0c13"
          },
          "source_id": "single-cell-omics-arena-soar-official-repository-resource",
          "source_type": "resource",
          "supports": [
            "/benchmark_version",
            "/scope",
            "/model_ids",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-08-13",
          "id": "soar-e5d2b3e-rna-zero-shot-cot-results-evidence",
          "locator": {
            "note": "Exact GPT-4o and GPT-4o mini values for the seven printed metrics.",
            "type": "repository-path",
            "value": "readme.md, SOAR-RNA Benchmark, Zero-shot Chain-of-thought Cell Type Annotation table at commit e5d2b3e2619cb56fece5fba78fae989a67fd0c13"
          },
          "source_id": "single-cell-omics-arena-soar-official-repository-resource",
          "source_type": "resource",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "soar-e5d2b3e-rna-zero-shot-cot",
      "metrics": [
        {
          "aggregation": "all predictions and normalized references",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "rouge-1",
          "pass_threshold": null,
          "range": null,
          "source_label": "R-1",
          "tolerance": null,
          "unit": "source-reported score"
        },
        {
          "aggregation": "all predictions and normalized references",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "rouge-2",
          "pass_threshold": null,
          "range": null,
          "source_label": "R-2",
          "tolerance": null,
          "unit": "source-reported score"
        },
        {
          "aggregation": "all predictions and normalized references",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "rouge-l",
          "pass_threshold": null,
          "range": null,
          "source_label": "R-L",
          "tolerance": null,
          "unit": "source-reported score"
        },
        {
          "aggregation": "all predictions and normalized references",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "meteor",
          "pass_threshold": null,
          "range": null,
          "source_label": "MET.",
          "tolerance": null,
          "unit": "source-reported score"
        },
        {
          "aggregation": "all predictions and normalized references; maximum order two",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "bleu-1",
          "pass_threshold": null,
          "range": null,
          "source_label": "B-1",
          "tolerance": null,
          "unit": "source-reported score"
        },
        {
          "aggregation": "all predictions and normalized references; maximum order two",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "bleu-2",
          "pass_threshold": null,
          "range": null,
          "source_label": "B-2",
          "tolerance": null,
          "unit": "source-reported score"
        },
        {
          "aggregation": "geometric average through order two over all predictions and normalized references",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "bleu",
          "pass_threshold": null,
          "range": null,
          "source_label": "BLEU",
          "tolerance": null,
          "unit": "source-reported score"
        }
      ],
      "model_ids": [
        "openai-gpt-4o-2024-05-13",
        "openai-gpt-4o-mini-2024-07-18"
      ],
      "protocol": {
        "contamination": {
          "notes": "No contamination or decontamination analysis is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": null,
          "model": null,
          "reporting_status": "reported",
          "type": "automatic evaluate-library text-overlap metrics"
        },
        "reasoning": {
          "notes": "The first prompt appends “Let's think step by step”; its response is inserted into the second prompt.",
          "reporting_status": "reported",
          "value": "zero-shot chain-of-thought"
        },
        "repeats": {
          "notes": "The published code traverses the artifact once, but the source does not label a formal evaluation repeat count.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "TaskConfig defines a local framework seed, but the OpenAI request passes no API seed.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "The official configuration selects the zero_shot_cot prompt template.",
          "reporting_status": "reported",
          "value": 0
        },
        "statistical": {
          "notes": "No confidence interval or repeated-run aggregation is reported.",
          "reporting_status": "reported",
          "value": "metrics computed over all predictions and normalized references"
        },
        "system_prompt_public": {
          "notes": "Both system/user prompt stages are published in the pinned repository.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "The OpenAI chat completion call explicitly sets temperature=0 for both calls.",
          "reporting_status": "reported",
          "value": 0
        },
        "time_budget": {
          "notes": "The pinned repository does not report a per-example time budget.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "GenerationConfig(max_new_tokens=1024) is forwarded as max_tokens for both calls.",
          "reporting_status": "reported",
          "value": "1024 maximum output tokens per call; two calls per example"
        },
        "tools": {
          "browser": {
            "notes": "The published ChatGPT pipeline passes only messages to the model and declares no tool interface.",
            "reporting_status": "reported",
            "value": false
          },
          "code_execution": {
            "notes": "The published ChatGPT pipeline passes only messages to the model and declares no tool interface.",
            "reporting_status": "reported",
            "value": false
          },
          "container": {
            "notes": "The published ChatGPT pipeline passes only messages to the model and declares no tool interface.",
            "reporting_status": "reported",
            "value": false
          },
          "databases": {
            "notes": "The published ChatGPT pipeline passes only messages to the model and declares no tool interface.",
            "reporting_status": "reported",
            "value": false
          },
          "external_tools": {
            "notes": "No external model tools are declared by the published pipeline.",
            "reporting_status": "reported",
            "value": []
          },
          "internet": {
            "notes": "The published ChatGPT pipeline passes only messages to the model and declares no tool interface.",
            "reporting_status": "reported",
            "value": false
          }
        },
        "turns": {
          "notes": "A reasoning response is generated first and inserted into a second direct-answer prompt.",
          "reporting_status": "reported",
          "value": "two model calls"
        }
      },
      "results": [
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "soar-e5d2b3e-rna-zero-shot-cot-results-evidence"
          ],
          "metric_id": "rouge-1",
          "model_id": "openai-gpt-4o-mini-2024-07-18",
          "n": 1191,
          "notes": "SOAR-RNA zero-shot CoT README table.",
          "status": "verified",
          "value": 51.63
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "soar-e5d2b3e-rna-zero-shot-cot-results-evidence"
          ],
          "metric_id": "rouge-2",
          "model_id": "openai-gpt-4o-mini-2024-07-18",
          "n": 1191,
          "notes": "SOAR-RNA zero-shot CoT README table.",
          "status": "verified",
          "value": 26.6
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "soar-e5d2b3e-rna-zero-shot-cot-results-evidence"
          ],
          "metric_id": "rouge-l",
          "model_id": "openai-gpt-4o-mini-2024-07-18",
          "n": 1191,
          "notes": "SOAR-RNA zero-shot CoT README table.",
          "status": "verified",
          "value": 51.17
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "soar-e5d2b3e-rna-zero-shot-cot-results-evidence"
          ],
          "metric_id": "meteor",
          "model_id": "openai-gpt-4o-mini-2024-07-18",
          "n": 1191,
          "notes": "SOAR-RNA zero-shot CoT README table.",
          "status": "verified",
          "value": 40.84
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "soar-e5d2b3e-rna-zero-shot-cot-results-evidence"
          ],
          "metric_id": "bleu-1",
          "model_id": "openai-gpt-4o-mini-2024-07-18",
          "n": 1191,
          "notes": "SOAR-RNA zero-shot CoT README table.",
          "status": "verified",
          "value": 50.29
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "soar-e5d2b3e-rna-zero-shot-cot-results-evidence"
          ],
          "metric_id": "bleu-2",
          "model_id": "openai-gpt-4o-mini-2024-07-18",
          "n": 1191,
          "notes": "SOAR-RNA zero-shot CoT README table.",
          "status": "verified",
          "value": 27.89
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "soar-e5d2b3e-rna-zero-shot-cot-results-evidence"
          ],
          "metric_id": "bleu",
          "model_id": "openai-gpt-4o-mini-2024-07-18",
          "n": 1191,
          "notes": "SOAR-RNA zero-shot CoT README table.",
          "status": "verified",
          "value": 37.45
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "soar-e5d2b3e-rna-zero-shot-cot-results-evidence"
          ],
          "metric_id": "rouge-1",
          "model_id": "openai-gpt-4o-2024-05-13",
          "n": 1191,
          "notes": "SOAR-RNA zero-shot CoT README table.",
          "status": "verified",
          "value": 57.67
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "soar-e5d2b3e-rna-zero-shot-cot-results-evidence"
          ],
          "metric_id": "rouge-2",
          "model_id": "openai-gpt-4o-2024-05-13",
          "n": 1191,
          "notes": "SOAR-RNA zero-shot CoT README table.",
          "status": "verified",
          "value": 31.55
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "soar-e5d2b3e-rna-zero-shot-cot-results-evidence"
          ],
          "metric_id": "rouge-l",
          "model_id": "openai-gpt-4o-2024-05-13",
          "n": 1191,
          "notes": "SOAR-RNA zero-shot CoT README table.",
          "status": "verified",
          "value": 57.34
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "soar-e5d2b3e-rna-zero-shot-cot-results-evidence"
          ],
          "metric_id": "meteor",
          "model_id": "openai-gpt-4o-2024-05-13",
          "n": 1191,
          "notes": "SOAR-RNA zero-shot CoT README table.",
          "status": "verified",
          "value": 45.36
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "soar-e5d2b3e-rna-zero-shot-cot-results-evidence"
          ],
          "metric_id": "bleu-1",
          "model_id": "openai-gpt-4o-2024-05-13",
          "n": 1191,
          "notes": "SOAR-RNA zero-shot CoT README table.",
          "status": "verified",
          "value": 55.27
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "soar-e5d2b3e-rna-zero-shot-cot-results-evidence"
          ],
          "metric_id": "bleu-2",
          "model_id": "openai-gpt-4o-2024-05-13",
          "n": 1191,
          "notes": "SOAR-RNA zero-shot CoT README table.",
          "status": "verified",
          "value": 32.15
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "soar-e5d2b3e-rna-zero-shot-cot-results-evidence"
          ],
          "metric_id": "bleu",
          "model_id": "openai-gpt-4o-2024-05-13",
          "n": 1191,
          "notes": "SOAR-RNA zero-shot CoT README table.",
          "status": "verified",
          "value": 42.15
        }
      ],
      "scope": {
        "filter": "All records in the pinned SOAR-RNA artifact, iterated in repository order with shuffle disabled.",
        "n": 1191,
        "reporting_status": "reported",
        "selection": "formal-subset",
        "subset_id": "soar-rna",
        "type": "subset"
      },
      "verification": {
        "last_verified": "2026-08-13",
        "notes": "Formal-subset two-call protocol and all 14 table values passed independent high-confidence verification.",
        "status": "verified"
      },
      "work_id": "single-cell-omics-arena-soar-repository-result-snapshot",
      "work_version_id": "single-cell-omics-arena-soar-repository-result-snapshot-e5d2b3e"
    },
    {
      "benchmark_id": "spatialbench",
      "benchmark_version": "paper-v2",
      "comparability_group": "spatialbench-paper-v2-base",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "spatialbench-paper-base-protocol-evidence",
          "locator": {
            "note": "Problem anatomy, deterministic graders, interactive compute, isolation, three repeats, 100-step limit, and two-stage statistics.",
            "type": "page",
            "value": "arXiv v2 §§3.2–3.7 and Appendix A.4"
          },
          "source_id": "spatialbench-preprint",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-22",
          "id": "spatialbench-paper-base-results-evidence",
          "locator": {
            "note": "Exact base-harness accuracy, steps, latency, cost, and 95% confidence intervals for seven source labels.",
            "type": "table",
            "value": "arXiv v2 Table 1"
          },
          "source_id": "spatialbench-preprint",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "spatialbench-paper-v2-base",
      "metrics": [
        {
          "aggregation": "mean over 146 evaluation-level means after three repeats",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Accuracy",
          "tolerance": "deterministic task-specific grader",
          "unit": "percent"
        },
        {
          "aggregation": "mean over evaluation-level three-run means",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "mean-steps",
          "pass_threshold": null,
          "range": [
            1,
            100
          ],
          "source_label": "Steps",
          "tolerance": null,
          "unit": "steps per evaluation"
        },
        {
          "aggregation": "mean over evaluation-level three-run means",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "mean-latency",
          "pass_threshold": null,
          "range": null,
          "source_label": "Latency",
          "tolerance": null,
          "unit": "seconds per evaluation"
        },
        {
          "aggregation": "mean over evaluation-level three-run means with missing cost logs excluded",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "mean-cost",
          "pass_threshold": null,
          "range": null,
          "source_label": "Cost",
          "tolerance": null,
          "unit": "USD per evaluation"
        }
      ],
      "model_ids": [
        "claude-opus-4-5",
        "claude-sonnet-4-5",
        "gemini-2-5-pro",
        "gpt-5-1",
        "gpt-5-2",
        "grok-4",
        "spatialbench-grok-4-1-unversioned"
      ],
      "protocol": {
        "contamination": {
          "notes": "No model-training decontamination study is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "five deterministic grader families"
        },
        "reasoning": {
          "notes": "Provider reasoning-effort controls are not reported.",
          "reporting_status": "reported",
          "value": "slightly modified Mini-SWE-Bench base harness"
        },
        "repeats": {
          "notes": "Every model configuration ran every evaluation three times.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Exact seed values are not reported.",
          "reporting_status": "reported",
          "value": "independent random seeds"
        },
        "shots": {
          "notes": "Demonstration count is not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "statistical": {
          "notes": "Missing/crashed evaluations count as failures; cost records missing from API logs are excluded only from cost aggregation.",
          "reporting_status": "reported",
          "value": "two-stage evaluation-weighted mean with t-distribution 95% confidence intervals over per-evaluation means"
        },
        "system_prompt_public": {
          "notes": "Task prompts and examples are printed, but the complete base-harness system prompt is not published in the paper.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "The paper documents the hard step limit but not the wall-clock value.",
          "reporting_status": "reported",
          "value": "maximum 100 agent steps; exact timeout not reported"
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Browser access is not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "code_execution": {
            "notes": "Agents interact with common scientific Python tooling and local data snapshots.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "Evaluations execute in isolated clean workspaces under fixed resource limits.",
            "reporting_status": "reported",
            "value": true
          },
          "databases": {
            "notes": "External database access is not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "external_tools": {
            "notes": "Harness-specific routing and prompts differ across conditions.",
            "reporting_status": "reported",
            "value": "interactive compute and local workspace"
          },
          "internet": {
            "notes": "Network access is not reported for the paper-era runs.",
            "reporting_status": "not_reported",
            "value": null
          }
        },
        "turns": {
          "notes": "A step is one model invocation plus resulting tool calls; up to 100 steps are allowed.",
          "reporting_status": "reported",
          "value": "multi-turn agent"
        }
      },
      "results": [
        {
          "ci_high": 45.44,
          "ci_low": 31.27,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-paper-base-results-evidence"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-opus-4-5",
          "n": 146,
          "notes": "Table 1",
          "status": "verified",
          "value": 38.36
        },
        {
          "ci_high": 3.17,
          "ci_low": 2.51,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-paper-base-results-evidence"
          ],
          "metric_id": "mean-steps",
          "model_id": "claude-opus-4-5",
          "n": 146,
          "notes": "Table 1",
          "status": "verified",
          "value": 2.84
        },
        {
          "ci_high": 142.9,
          "ci_low": 104.8,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-paper-base-results-evidence"
          ],
          "metric_id": "mean-latency",
          "model_id": "claude-opus-4-5",
          "n": 146,
          "notes": "Table 1",
          "status": "verified",
          "value": 123.8
        },
        {
          "ci_high": 0.165,
          "ci_low": 0.121,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-paper-base-results-evidence"
          ],
          "metric_id": "mean-cost",
          "model_id": "claude-opus-4-5",
          "n": 146,
          "notes": "Table 1",
          "status": "verified",
          "value": 0.143
        },
        {
          "ci_high": 34.4,
          "ci_low": 22.22,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-paper-base-results-evidence"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-sonnet-4-5",
          "n": 146,
          "notes": "Table 1",
          "status": "verified",
          "value": 28.31
        },
        {
          "ci_high": 2.7,
          "ci_low": 2.17,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-paper-base-results-evidence"
          ],
          "metric_id": "mean-steps",
          "model_id": "claude-sonnet-4-5",
          "n": 146,
          "notes": "Table 1",
          "status": "verified",
          "value": 2.43
        },
        {
          "ci_high": 133.2,
          "ci_low": 98.0,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-paper-base-results-evidence"
          ],
          "metric_id": "mean-latency",
          "model_id": "claude-sonnet-4-5",
          "n": 146,
          "notes": "Table 1",
          "status": "verified",
          "value": 115.6
        },
        {
          "ci_high": 0.093,
          "ci_low": 0.068,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-paper-base-results-evidence"
          ],
          "metric_id": "mean-cost",
          "model_id": "claude-sonnet-4-5",
          "n": 146,
          "notes": "Table 1",
          "status": "verified",
          "value": 0.081
        },
        {
          "ci_high": 40.47,
          "ci_low": 27.57,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-paper-base-results-evidence"
          ],
          "metric_id": "accuracy",
          "model_id": "gpt-5-2",
          "n": 146,
          "notes": "Table 1",
          "status": "verified",
          "value": 34.02
        },
        {
          "ci_high": 2.3,
          "ci_low": 1.89,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-paper-base-results-evidence"
          ],
          "metric_id": "mean-steps",
          "model_id": "gpt-5-2",
          "n": 146,
          "notes": "Table 1",
          "status": "verified",
          "value": 2.1
        },
        {
          "ci_high": 101.4,
          "ci_low": 76.9,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-paper-base-results-evidence"
          ],
          "metric_id": "mean-latency",
          "model_id": "gpt-5-2",
          "n": 146,
          "notes": "Table 1",
          "status": "verified",
          "value": 89.2
        },
        {
          "ci_high": 0.042,
          "ci_low": 0.031,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-paper-base-results-evidence"
          ],
          "metric_id": "mean-cost",
          "model_id": "gpt-5-2",
          "n": 146,
          "notes": "Table 1",
          "status": "verified",
          "value": 0.037
        },
        {
          "ci_high": 33.5,
          "ci_low": 21.3,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-paper-base-results-evidence"
          ],
          "metric_id": "accuracy",
          "model_id": "gpt-5-1",
          "n": 146,
          "notes": "Table 1",
          "status": "verified",
          "value": 27.4
        },
        {
          "ci_high": 2.58,
          "ci_low": 2.18,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-paper-base-results-evidence"
          ],
          "metric_id": "mean-steps",
          "model_id": "gpt-5-1",
          "n": 146,
          "notes": "Table 1",
          "status": "verified",
          "value": 2.38
        },
        {
          "ci_high": 63.5,
          "ci_low": 48.2,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-paper-base-results-evidence"
          ],
          "metric_id": "mean-latency",
          "model_id": "gpt-5-1",
          "n": 146,
          "notes": "Table 1",
          "status": "verified",
          "value": 55.8
        },
        {
          "ci_high": 0.023,
          "ci_low": 0.017,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-paper-base-results-evidence"
          ],
          "metric_id": "mean-cost",
          "model_id": "gpt-5-1",
          "n": 146,
          "notes": "Table 1",
          "status": "verified",
          "value": 0.02
        },
        {
          "ci_high": 25.43,
          "ci_low": 14.75,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-paper-base-results-evidence"
          ],
          "metric_id": "accuracy",
          "model_id": "gemini-2-5-pro",
          "n": 146,
          "notes": "Table 1",
          "status": "verified",
          "value": 20.09
        },
        {
          "ci_high": 4.1,
          "ci_low": 3.11,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-paper-base-results-evidence"
          ],
          "metric_id": "mean-steps",
          "model_id": "gemini-2-5-pro",
          "n": 146,
          "notes": "Table 1",
          "status": "verified",
          "value": 3.61
        },
        {
          "ci_high": 218.7,
          "ci_low": 168.2,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-paper-base-results-evidence"
          ],
          "metric_id": "mean-latency",
          "model_id": "gemini-2-5-pro",
          "n": 146,
          "notes": "Table 1",
          "status": "verified",
          "value": 193.5
        },
        {
          "ci_high": 0.219,
          "ci_low": 0.157,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-paper-base-results-evidence"
          ],
          "metric_id": "mean-cost",
          "model_id": "gemini-2-5-pro",
          "n": 146,
          "notes": "Table 1",
          "status": "verified",
          "value": 0.188
        },
        {
          "ci_high": 27.98,
          "ci_low": 17.68,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-paper-base-results-evidence"
          ],
          "metric_id": "accuracy",
          "model_id": "grok-4",
          "n": 146,
          "notes": "Table 1",
          "status": "verified",
          "value": 22.83
        },
        {
          "ci_high": 12.11,
          "ci_low": 7.68,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-paper-base-results-evidence"
          ],
          "metric_id": "mean-steps",
          "model_id": "grok-4",
          "n": 146,
          "notes": "Table 1",
          "status": "verified",
          "value": 9.9
        },
        {
          "ci_high": 198.3,
          "ci_low": 148.0,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-paper-base-results-evidence"
          ],
          "metric_id": "mean-latency",
          "model_id": "grok-4",
          "n": 146,
          "notes": "Table 1",
          "status": "verified",
          "value": 173.2
        },
        {
          "ci_high": 0.067,
          "ci_low": 0.03,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-paper-base-results-evidence"
          ],
          "metric_id": "mean-cost",
          "model_id": "grok-4",
          "n": 146,
          "notes": "Table 1",
          "status": "verified",
          "value": 0.048
        },
        {
          "ci_high": 29.54,
          "ci_low": 19.78,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-paper-base-results-evidence"
          ],
          "metric_id": "accuracy",
          "model_id": "spatialbench-grok-4-1-unversioned",
          "n": 146,
          "notes": "Table 1",
          "status": "verified",
          "value": 24.66
        },
        {
          "ci_high": 11.78,
          "ci_low": 8.09,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-paper-base-results-evidence"
          ],
          "metric_id": "mean-steps",
          "model_id": "spatialbench-grok-4-1-unversioned",
          "n": 146,
          "notes": "Table 1",
          "status": "verified",
          "value": 9.93
        },
        {
          "ci_high": 217.3,
          "ci_low": 175.4,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-paper-base-results-evidence"
          ],
          "metric_id": "mean-latency",
          "model_id": "spatialbench-grok-4-1-unversioned",
          "n": 146,
          "notes": "Table 1",
          "status": "verified",
          "value": 196.4
        },
        {
          "ci_high": 0.107,
          "ci_low": 0.047,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-paper-base-results-evidence"
          ],
          "metric_id": "mean-cost",
          "model_id": "spatialbench-grok-4-1-unversioned",
          "n": 146,
          "notes": "Table 1",
          "status": "verified",
          "value": 0.077
        }
      ],
      "scope": {
        "filter": null,
        "n": 146,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Full 146-evaluation base-harness results are transcribed only from labeled Table 1 values.",
        "status": "verified"
      },
      "work_id": "spatialbench-preprint",
      "work_version_id": "spatialbench-preprint-v2"
    },
    {
      "benchmark_id": "spatialbench",
      "benchmark_version": "paper-v2",
      "comparability_group": "spatialbench-paper-v2-claude-code",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "spatialbench-paper-claude-code-protocol-evidence",
          "locator": {
            "note": "Harness separation, full scope, deterministic graders, three repeats, and statistics.",
            "type": "page",
            "value": "arXiv v2 §§2.5 and 3.5–3.7; Appendix A.4"
          },
          "source_id": "spatialbench-preprint",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-22",
          "id": "spatialbench-paper-claude-code-results-evidence",
          "locator": {
            "note": "Exact labeled Claude Code accuracy and 95% confidence intervals.",
            "type": "table",
            "value": "arXiv v2 Table 4"
          },
          "source_id": "spatialbench-preprint",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "spatialbench-paper-v2-claude-code",
      "metrics": [
        {
          "aggregation": "mean over 146 evaluation-level means after three repeats",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Accuracy",
          "tolerance": "deterministic task-specific grader",
          "unit": "percent"
        }
      ],
      "model_ids": [
        "claude-opus-4-5",
        "claude-sonnet-4-5"
      ],
      "protocol": {
        "contamination": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "five deterministic grader families"
        },
        "reasoning": {
          "notes": "Provider effort control is not reported.",
          "reporting_status": "reported",
          "value": "Claude Code harness"
        },
        "repeats": {
          "notes": "Three independent runs per evaluation.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Exact values not reported.",
          "reporting_status": "reported",
          "value": "independent random seeds"
        },
        "shots": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "statistical": {
          "notes": "Evaluation is the statistical unit.",
          "reporting_status": "reported",
          "value": "two-stage evaluation-weighted mean with t-distribution 95% confidence intervals"
        },
        "system_prompt_public": {
          "notes": "Harness behavior is described but the complete prompt is not printed.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "The paper states budgets were consistent within harness.",
          "reporting_status": "reported",
          "value": "fixed harness-specific budget; numeric value not reported"
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "code_execution": {
            "notes": "Common scientific Python tooling and local data workspace.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "Isolated evaluation workspace.",
            "reporting_status": "reported",
            "value": true
          },
          "databases": {
            "notes": "Not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "external_tools": {
            "notes": "Exact tool routing is not enumerated.",
            "reporting_status": "reported",
            "value": "Claude Code harness tools"
          },
          "internet": {
            "notes": "Not reported.",
            "reporting_status": "not_reported",
            "value": null
          }
        },
        "turns": {
          "notes": "Interactive agent harness.",
          "reporting_status": "reported",
          "value": "multi-turn Claude Code agent"
        }
      },
      "results": [
        {
          "ci_high": 55.3,
          "ci_low": 40.9,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-paper-claude-code-results-evidence"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-opus-4-5",
          "n": 146,
          "notes": "Table 4",
          "status": "verified",
          "value": 48.1
        },
        {
          "ci_high": 52.2,
          "ci_low": 38.0,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-paper-claude-code-results-evidence"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-sonnet-4-5",
          "n": 146,
          "notes": "Table 4",
          "status": "verified",
          "value": 45.1
        }
      ],
      "scope": {
        "filter": null,
        "n": 146,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Claude Code is isolated from base and Latch harness comparisons.",
        "status": "verified"
      },
      "work_id": "spatialbench-preprint",
      "work_version_id": "spatialbench-preprint-v2"
    },
    {
      "benchmark_id": "spatialbench",
      "benchmark_version": "paper-v2",
      "comparability_group": "spatialbench-paper-v2-latch",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "spatialbench-paper-latch-protocol-evidence",
          "locator": {
            "note": "Harness separation, full scope, deterministic graders, three repeats, and statistics.",
            "type": "page",
            "value": "arXiv v2 §§2.5 and 3.5–3.7; Appendix A.4"
          },
          "source_id": "spatialbench-preprint",
          "source_type": "work",
          "supports": [
            "/benchmark_version",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-22",
          "id": "spatialbench-paper-latch-results-evidence",
          "locator": {
            "note": "Exact labeled Latch accuracy and 95% confidence interval.",
            "type": "table",
            "value": "arXiv v2 Table 4"
          },
          "source_id": "spatialbench-preprint",
          "source_type": "work",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "spatialbench-paper-v2-latch",
      "metrics": [
        {
          "aggregation": "mean over 146 evaluation-level means after three repeats",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Accuracy",
          "tolerance": "deterministic task-specific grader",
          "unit": "percent"
        }
      ],
      "model_ids": [
        "claude-opus-4-5"
      ],
      "protocol": {
        "contamination": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "five deterministic grader families"
        },
        "reasoning": {
          "notes": "Provider effort control is not reported.",
          "reporting_status": "reported",
          "value": "Latch agent harness"
        },
        "repeats": {
          "notes": "Three independent runs per evaluation.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Exact values not reported.",
          "reporting_status": "reported",
          "value": "independent random seeds"
        },
        "shots": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "statistical": {
          "notes": "Evaluation is the statistical unit.",
          "reporting_status": "reported",
          "value": "two-stage evaluation-weighted mean with t-distribution 95% confidence intervals"
        },
        "system_prompt_public": {
          "notes": "The complete Latch harness prompt is not printed.",
          "reporting_status": "reported",
          "value": false
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "The paper states budgets were consistent within harness.",
          "reporting_status": "reported",
          "value": "fixed harness-specific budget; numeric value not reported"
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "code_execution": {
            "notes": "Common scientific Python tooling and local data workspace.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "Isolated evaluation workspace.",
            "reporting_status": "reported",
            "value": true
          },
          "databases": {
            "notes": "Not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "external_tools": {
            "notes": "Exact tool routing is not enumerated.",
            "reporting_status": "reported",
            "value": "Latch agent harness tools"
          },
          "internet": {
            "notes": "Not reported.",
            "reporting_status": "not_reported",
            "value": null
          }
        },
        "turns": {
          "notes": "Interactive agent harness.",
          "reporting_status": "reported",
          "value": "multi-turn Latch agent"
        }
      },
      "results": [
        {
          "ci_high": 68.1,
          "ci_low": 55.3,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-paper-latch-results-evidence"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-opus-4-5",
          "n": 146,
          "notes": "Table 4",
          "status": "verified",
          "value": 61.7
        }
      ],
      "scope": {
        "filter": null,
        "n": 146,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Latch harness result is isolated from base and Claude Code settings.",
        "status": "verified"
      },
      "work_id": "spatialbench-preprint",
      "work_version_id": "spatialbench-preprint-v2"
    },
    {
      "benchmark_id": "spatialbench",
      "benchmark_version": "repo-159-5042c4f",
      "comparability_group": "spatialbench-repo-159-claude-code",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "spatialbench-repo-claude-code-protocol-evidence",
          "locator": {
            "note": "Three runs, max reasoning, container resources, six-hour timeout, network access, binary placement, and aggregation.",
            "type": "repository-path",
            "value": "METHODS.md at commit 5042c4f3ee597da1590650c7b894d068ae968e26"
          },
          "source_id": "spatialbench-repository-resource",
          "source_type": "resource",
          "supports": [
            "/benchmark_version",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-22",
          "id": "spatialbench-repo-claude-code-results-evidence",
          "locator": {
            "note": "Exact model label, accuracy and CI, mean cost, mean duration, and N.",
            "type": "repository-path",
            "value": "results/model_results.csv at commit 5042c4f3ee597da1590650c7b894d068ae968e26; harness=claude-code"
          },
          "source_id": "spatialbench-repository-resource",
          "source_type": "resource",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "spatialbench-repo-159-claude-code",
      "metrics": [
        {
          "aggregation": "mean over 159 evaluation-level means after three repeats",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Accuracy",
          "tolerance": "deterministic task-specific grader",
          "unit": "percent"
        },
        {
          "aggregation": "mean over the full result set",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "mean-cost",
          "pass_threshold": null,
          "range": null,
          "source_label": "Cost",
          "tolerance": null,
          "unit": "USD per evaluation"
        },
        {
          "aggregation": "mean over the full result set",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "mean-duration",
          "pass_threshold": null,
          "range": null,
          "source_label": "Duration",
          "tolerance": null,
          "unit": "seconds per evaluation"
        }
      ],
      "model_ids": [
        "claude-opus-4-7",
        "claude-opus-4-8"
      ],
      "protocol": {
        "contamination": {
          "notes": "Only a representative sample is public.",
          "reporting_status": "reported",
          "value": "full suite withheld to reduce contamination"
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "five deterministic grader families"
        },
        "reasoning": {
          "notes": "Model-specific effort strings are not enumerated in the result CSV.",
          "reporting_status": "reported",
          "value": "maximum reasoning effort where applicable"
        },
        "repeats": {
          "notes": "Every evaluation is run three times and averaged.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Trial identifiers are public but numeric seeds are not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Demonstration count is not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "statistical": {
          "notes": "N=159 in the official model result CSV.",
          "reporting_status": "reported",
          "value": "t-distribution 95% confidence interval over per-evaluation means after averaging three runs"
        },
        "system_prompt_public": {
          "notes": "The official methods link the public latch-eval-tools configuration.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "OOM events trigger a restart with a warning.",
          "reporting_status": "reported",
          "value": "21600 seconds per task with no step limit"
        },
        "token_budget": {
          "notes": "No token limit is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Browser availability is not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "code_execution": {
            "notes": "Claude Code executes inside the evaluation container.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "4 vCPU, 32 GB RAM, 100 GB storage container.",
            "reporting_status": "reported",
            "value": true
          },
          "databases": {
            "notes": "No fixed database suite is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "external_tools": {
            "notes": "Harness-specific public configuration is linked from METHODS.md.",
            "reporting_status": "reported",
            "value": "Claude Code with common scientific libraries and network-enabled shell"
          },
          "internet": {
            "notes": "Container network access is enabled; downloads and searches are discouraged but not blocked.",
            "reporting_status": "reported",
            "value": true
          }
        },
        "turns": {
          "notes": "Binary runs inside the evaluation container.",
          "reporting_status": "reported",
          "value": "multi-turn Claude Code"
        }
      },
      "results": [
        {
          "ci_high": 62.47,
          "ci_low": 48.22,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-claude-code-results-evidence"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-opus-4-8",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 55.35
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-claude-code-results-evidence"
          ],
          "metric_id": "mean-cost",
          "model_id": "claude-opus-4-8",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 0.8776
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-claude-code-results-evidence"
          ],
          "metric_id": "mean-duration",
          "model_id": "claude-opus-4-8",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 489.16
        },
        {
          "ci_high": 58.57,
          "ci_low": 44.15,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-claude-code-results-evidence"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-opus-4-7",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 51.36
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-claude-code-results-evidence"
          ],
          "metric_id": "mean-cost",
          "model_id": "claude-opus-4-7",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 0.8023
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-claude-code-results-evidence"
          ],
          "metric_id": "mean-duration",
          "model_id": "claude-opus-4-7",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 532.85
        }
      ],
      "scope": {
        "filter": null,
        "n": 159,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Exact 159-evaluation Claude Code rows transcribed from the pinned official CSV.",
        "status": "verified"
      },
      "work_id": "spatialbench-repository-release",
      "work_version_id": "spatialbench-repository-release-2026-06-10"
    },
    {
      "benchmark_id": "spatialbench",
      "benchmark_version": "repo-159-5042c4f",
      "comparability_group": "spatialbench-repo-159-mini-swe-agent",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "spatialbench-repo-mini-protocol-evidence",
          "locator": {
            "note": "Three runs, max reasoning, container resources, six-hour timeout, network access, harness placement, and aggregation.",
            "type": "repository-path",
            "value": "METHODS.md at commit 5042c4f3ee597da1590650c7b894d068ae968e26"
          },
          "source_id": "spatialbench-repository-resource",
          "source_type": "resource",
          "supports": [
            "/benchmark_version",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-22",
          "id": "spatialbench-repo-mini-results-evidence",
          "locator": {
            "note": "Exact model label, accuracy and CI, mean cost, mean duration, and N.",
            "type": "repository-path",
            "value": "results/model_results.csv at commit 5042c4f3ee597da1590650c7b894d068ae968e26; harness=mini-swe-agent"
          },
          "source_id": "spatialbench-repository-resource",
          "source_type": "resource",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "spatialbench-repo-159-mini-swe-agent",
      "metrics": [
        {
          "aggregation": "mean over 159 evaluation-level means after three repeats",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Accuracy",
          "tolerance": "deterministic task-specific grader",
          "unit": "percent"
        },
        {
          "aggregation": "mean over the full result set",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "mean-cost",
          "pass_threshold": null,
          "range": null,
          "source_label": "Cost",
          "tolerance": null,
          "unit": "USD per evaluation"
        },
        {
          "aggregation": "mean over the full result set",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "mean-duration",
          "pass_threshold": null,
          "range": null,
          "source_label": "Duration",
          "tolerance": null,
          "unit": "seconds per evaluation"
        }
      ],
      "model_ids": [
        "claude-opus-4-5",
        "claude-opus-4-6",
        "claude-opus-4-7",
        "claude-opus-4-8",
        "claude-sonnet-4-5",
        "claude-sonnet-4-6",
        "gemini-2-5-pro",
        "gemini-3-1-pro-preview",
        "gemini-3-5-flash",
        "gpt-5-1",
        "gpt-5-2",
        "gpt-5-4",
        "gpt-5-5",
        "grok-4",
        "grok-4-1-fast-reasoning",
        "grok-4-20-beta-0309-reasoning"
      ],
      "protocol": {
        "contamination": {
          "notes": "Only a representative sample is public.",
          "reporting_status": "reported",
          "value": "full suite withheld to reduce contamination"
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "five deterministic grader families"
        },
        "reasoning": {
          "notes": "Model-specific effort strings are not enumerated in the result CSV.",
          "reporting_status": "reported",
          "value": "maximum reasoning effort where applicable"
        },
        "repeats": {
          "notes": "Every evaluation is run three times and averaged.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Trial identifiers are public but numeric seeds are not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Demonstration count is not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "statistical": {
          "notes": "N=159 in the official model result CSV.",
          "reporting_status": "reported",
          "value": "t-distribution 95% confidence interval over per-evaluation means after averaging three runs"
        },
        "system_prompt_public": {
          "notes": "The official methods link the public latch-eval-tools configuration.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "OOM events trigger a restart with a warning.",
          "reporting_status": "reported",
          "value": "21600 seconds per task with no step limit"
        },
        "token_budget": {
          "notes": "No token limit is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Browser availability is not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "code_execution": {
            "notes": "Tool calls execute inside the evaluation container.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "4 vCPU, 32 GB RAM, 100 GB storage container.",
            "reporting_status": "reported",
            "value": true
          },
          "databases": {
            "notes": "No fixed database suite is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "external_tools": {
            "notes": "Harness-specific public configuration is linked from METHODS.md.",
            "reporting_status": "reported",
            "value": "mini-swe-agent with common scientific libraries and network-enabled shell"
          },
          "internet": {
            "notes": "Container network access is enabled; downloads and searches are discouraged but not blocked.",
            "reporting_status": "reported",
            "value": true
          }
        },
        "turns": {
          "notes": "Harness runs outside the task container and executes tool calls inside it.",
          "reporting_status": "reported",
          "value": "multi-turn mini-swe-agent"
        }
      },
      "results": [
        {
          "ci_high": 64.59,
          "ci_low": 50.71,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "accuracy",
          "model_id": "gpt-5-5",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 57.65
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "mean-cost",
          "model_id": "gpt-5-5",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 1.1207
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "mean-duration",
          "model_id": "gpt-5-5",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 586.66
        },
        {
          "ci_high": 64.16,
          "ci_low": 50.72,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "accuracy",
          "model_id": "gpt-5-4",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 57.44
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "mean-cost",
          "model_id": "gpt-5-4",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 0.577
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "mean-duration",
          "model_id": "gpt-5-4",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 1128.75
        },
        {
          "ci_high": 59.94,
          "ci_low": 45.72,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-opus-4-6",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 52.83
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "mean-cost",
          "model_id": "claude-opus-4-6",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 0.8456
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "mean-duration",
          "model_id": "claude-opus-4-6",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 609.15
        },
        {
          "ci_high": 59.75,
          "ci_low": 45.49,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-opus-4-8",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 52.62
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "mean-cost",
          "model_id": "claude-opus-4-8",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 1.1061
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "mean-duration",
          "model_id": "claude-opus-4-8",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 902.86
        },
        {
          "ci_high": 59.54,
          "ci_low": 45.28,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-opus-4-7",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 52.41
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "mean-cost",
          "model_id": "claude-opus-4-7",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 0.9817
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "mean-duration",
          "model_id": "claude-opus-4-7",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 626.84
        },
        {
          "ci_high": 58.52,
          "ci_low": 44.63,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "accuracy",
          "model_id": "gemini-3-1-pro-preview",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 51.57
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "mean-cost",
          "model_id": "gemini-3-1-pro-preview",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 0.9362
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "mean-duration",
          "model_id": "gemini-3-1-pro-preview",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 1061.63
        },
        {
          "ci_high": 57.25,
          "ci_low": 42.96,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "accuracy",
          "model_id": "gpt-5-2",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 50.1
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "mean-cost",
          "model_id": "gpt-5-2",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 0.6024
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "mean-duration",
          "model_id": "gpt-5-2",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 931.76
        },
        {
          "ci_high": 55.56,
          "ci_low": 42.14,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "accuracy",
          "model_id": "gemini-3-5-flash",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 48.85
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "mean-cost",
          "model_id": "gemini-3-5-flash",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 2.7608
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "mean-duration",
          "model_id": "gemini-3-5-flash",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 1145.4
        },
        {
          "ci_high": 52.5,
          "ci_low": 39.33,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "accuracy",
          "model_id": "grok-4-20-beta-0309-reasoning",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 45.91
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "mean-cost",
          "model_id": "grok-4-20-beta-0309-reasoning",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 0.1679
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "mean-duration",
          "model_id": "grok-4-20-beta-0309-reasoning",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 342.63
        },
        {
          "ci_high": 51.12,
          "ci_low": 37.35,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-sonnet-4-6",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 44.23
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "mean-cost",
          "model_id": "claude-sonnet-4-6",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 0.273
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "mean-duration",
          "model_id": "claude-sonnet-4-6",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 405.3
        },
        {
          "ci_high": 49.42,
          "ci_low": 36.12,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-opus-4-5",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 42.77
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "mean-cost",
          "model_id": "claude-opus-4-5",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 0.4624
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "mean-duration",
          "model_id": "claude-opus-4-5",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 376.54
        },
        {
          "ci_high": 48.1,
          "ci_low": 34.92,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "accuracy",
          "model_id": "claude-sonnet-4-5",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 41.51
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "mean-cost",
          "model_id": "claude-sonnet-4-5",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 0.2247
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "mean-duration",
          "model_id": "claude-sonnet-4-5",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 294.44
        },
        {
          "ci_high": 46.31,
          "ci_low": 33.36,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "accuracy",
          "model_id": "gpt-5-1",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 39.83
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "mean-cost",
          "model_id": "gpt-5-1",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 0.1574
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "mean-duration",
          "model_id": "gpt-5-1",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 309.79
        },
        {
          "ci_high": 40.22,
          "ci_low": 27.7,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "accuracy",
          "model_id": "grok-4-1-fast-reasoning",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 33.96
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "mean-cost",
          "model_id": "grok-4-1-fast-reasoning",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 0.0164
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "mean-duration",
          "model_id": "grok-4-1-fast-reasoning",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 357.0
        },
        {
          "ci_high": 37.48,
          "ci_low": 26.25,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "accuracy",
          "model_id": "grok-4",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 31.87
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "mean-cost",
          "model_id": "grok-4",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 0.4529
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "mean-duration",
          "model_id": "grok-4",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 732.47
        },
        {
          "ci_high": 34.9,
          "ci_low": 22.97,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "accuracy",
          "model_id": "gemini-2-5-pro",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 28.93
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "mean-cost",
          "model_id": "gemini-2-5-pro",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 0.1086
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-mini-results-evidence"
          ],
          "metric_id": "mean-duration",
          "model_id": "gemini-2-5-pro",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 231.07
        }
      ],
      "scope": {
        "filter": null,
        "n": 159,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Exact 159-evaluation mini-swe-agent rows transcribed from the pinned official CSV.",
        "status": "verified"
      },
      "work_id": "spatialbench-repository-release",
      "work_version_id": "spatialbench-repository-release-2026-06-10"
    },
    {
      "benchmark_id": "spatialbench",
      "benchmark_version": "repo-159-5042c4f",
      "comparability_group": "spatialbench-repo-159-openai-codex",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "spatialbench-repo-openai-codex-protocol-evidence",
          "locator": {
            "note": "Three runs, max reasoning, container resources, six-hour timeout, network access, binary placement, and aggregation.",
            "type": "repository-path",
            "value": "METHODS.md at commit 5042c4f3ee597da1590650c7b894d068ae968e26"
          },
          "source_id": "spatialbench-repository-resource",
          "source_type": "resource",
          "supports": [
            "/benchmark_version",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-22",
          "id": "spatialbench-repo-openai-codex-results-evidence",
          "locator": {
            "note": "Exact model label, accuracy and CI, mean cost, mean duration, and N.",
            "type": "repository-path",
            "value": "results/model_results.csv at commit 5042c4f3ee597da1590650c7b894d068ae968e26; harness=openai-codex"
          },
          "source_id": "spatialbench-repository-resource",
          "source_type": "resource",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "spatialbench-repo-159-openai-codex",
      "metrics": [
        {
          "aggregation": "mean over 159 evaluation-level means after three repeats",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Accuracy",
          "tolerance": "deterministic task-specific grader",
          "unit": "percent"
        },
        {
          "aggregation": "mean over the full result set",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "mean-cost",
          "pass_threshold": null,
          "range": null,
          "source_label": "Cost",
          "tolerance": null,
          "unit": "USD per evaluation"
        },
        {
          "aggregation": "mean over the full result set",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "mean-duration",
          "pass_threshold": null,
          "range": null,
          "source_label": "Duration",
          "tolerance": null,
          "unit": "seconds per evaluation"
        }
      ],
      "model_ids": [
        "gpt-5-5"
      ],
      "protocol": {
        "contamination": {
          "notes": "Only a representative sample is public.",
          "reporting_status": "reported",
          "value": "full suite withheld to reduce contamination"
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "five deterministic grader families"
        },
        "reasoning": {
          "notes": "Exact Codex-specific effort string is not reported.",
          "reporting_status": "reported",
          "value": "maximum reasoning effort where applicable"
        },
        "repeats": {
          "notes": "Every evaluation is run three times and averaged.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Trial identifiers are public but numeric seeds are not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Demonstration count is not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "statistical": {
          "notes": "N=159 in the official model result CSV.",
          "reporting_status": "reported",
          "value": "t-distribution 95% confidence interval over per-evaluation means after averaging three runs"
        },
        "system_prompt_public": {
          "notes": "The official methods link the public latch-eval-tools configuration.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "OOM events trigger a restart with a warning.",
          "reporting_status": "reported",
          "value": "21600 seconds per task with no step limit"
        },
        "token_budget": {
          "notes": "No token limit is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Browser availability is not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "code_execution": {
            "notes": "OpenAI Codex executes inside the evaluation container.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "4 vCPU, 32 GB RAM, 100 GB storage container.",
            "reporting_status": "reported",
            "value": true
          },
          "databases": {
            "notes": "No fixed database suite is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "external_tools": {
            "notes": "Public configuration is linked from METHODS.md.",
            "reporting_status": "reported",
            "value": "OpenAI Codex with common scientific libraries and network-enabled shell"
          },
          "internet": {
            "notes": "Container network access is enabled; downloads and searches are discouraged but not blocked.",
            "reporting_status": "reported",
            "value": true
          }
        },
        "turns": {
          "notes": "Binary runs inside the evaluation container.",
          "reporting_status": "reported",
          "value": "multi-turn OpenAI Codex"
        }
      },
      "results": [
        {
          "ci_high": 60.72,
          "ci_low": 46.62,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-openai-codex-results-evidence"
          ],
          "metric_id": "accuracy",
          "model_id": "gpt-5-5",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 53.67
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-openai-codex-results-evidence"
          ],
          "metric_id": "mean-cost",
          "model_id": "gpt-5-5",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 3.1616
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-openai-codex-results-evidence"
          ],
          "metric_id": "mean-duration",
          "model_id": "gpt-5-5",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 382.01
        }
      ],
      "scope": {
        "filter": null,
        "n": 159,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Exact 159-evaluation OpenAI Codex row transcribed from the pinned official CSV.",
        "status": "verified"
      },
      "work_id": "spatialbench-repository-release",
      "work_version_id": "spatialbench-repository-release-2026-06-10"
    },
    {
      "benchmark_id": "spatialbench",
      "benchmark_version": "repo-159-5042c4f",
      "comparability_group": "spatialbench-repo-159-pi",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "spatialbench-repo-pi-protocol-evidence",
          "locator": {
            "note": "Three runs, max reasoning, container resources, six-hour timeout, network access, and aggregation.",
            "type": "repository-path",
            "value": "METHODS.md at commit 5042c4f3ee597da1590650c7b894d068ae968e26"
          },
          "source_id": "spatialbench-repository-resource",
          "source_type": "resource",
          "supports": [
            "/benchmark_version",
            "/scope",
            "/protocol",
            "/metrics"
          ]
        },
        {
          "accessed_date": "2026-07-22",
          "id": "spatialbench-repo-pi-results-evidence",
          "locator": {
            "note": "Exact model label, accuracy and CI, mean cost, mean duration, and N.",
            "type": "repository-path",
            "value": "results/model_results.csv at commit 5042c4f3ee597da1590650c7b894d068ae968e26; harness=pi"
          },
          "source_id": "spatialbench-repository-resource",
          "source_type": "resource",
          "supports": [
            "/results"
          ]
        }
      ],
      "id": "spatialbench-repo-159-pi",
      "metrics": [
        {
          "aggregation": "mean over 159 evaluation-level means after three repeats",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Accuracy",
          "tolerance": "deterministic task-specific grader",
          "unit": "percent"
        },
        {
          "aggregation": "mean over the full result set",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "mean-cost",
          "pass_threshold": null,
          "range": null,
          "source_label": "Cost",
          "tolerance": null,
          "unit": "USD per evaluation"
        },
        {
          "aggregation": "mean over the full result set",
          "baseline_model_id": null,
          "higher_is_better": false,
          "kind": "absolute",
          "metric_id": "mean-duration",
          "pass_threshold": null,
          "range": null,
          "source_label": "Duration",
          "tolerance": null,
          "unit": "seconds per evaluation"
        }
      ],
      "model_ids": [
        "gemini-3-5-flash"
      ],
      "protocol": {
        "contamination": {
          "notes": "Only a representative sample is public.",
          "reporting_status": "reported",
          "value": "full suite withheld to reduce contamination"
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "five deterministic grader families"
        },
        "reasoning": {
          "notes": "Exact Pi-specific effort string is not reported.",
          "reporting_status": "reported",
          "value": "maximum reasoning effort where applicable"
        },
        "repeats": {
          "notes": "Every evaluation is run three times and averaged.",
          "reporting_status": "reported",
          "value": 3
        },
        "seed": {
          "notes": "Trial identifiers are public but numeric seeds are not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Demonstration count is not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "statistical": {
          "notes": "N=159 in the official model result CSV.",
          "reporting_status": "reported",
          "value": "t-distribution 95% confidence interval over per-evaluation means after averaging three runs"
        },
        "system_prompt_public": {
          "notes": "The official methods link the public latch-eval-tools configuration.",
          "reporting_status": "reported",
          "value": true
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "OOM events trigger a restart with a warning.",
          "reporting_status": "reported",
          "value": "21600 seconds per task with no step limit"
        },
        "token_budget": {
          "notes": "No token limit is reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Browser availability is not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "code_execution": {
            "notes": "Tool calls execute against the evaluation container.",
            "reporting_status": "reported",
            "value": true
          },
          "container": {
            "notes": "4 vCPU, 32 GB RAM, 100 GB storage container.",
            "reporting_status": "reported",
            "value": true
          },
          "databases": {
            "notes": "No fixed database suite is reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "external_tools": {
            "notes": "Public configuration is linked from METHODS.md.",
            "reporting_status": "reported",
            "value": "Pi harness with common scientific libraries and network-enabled shell"
          },
          "internet": {
            "notes": "Container network access is enabled; downloads and searches are discouraged but not blocked.",
            "reporting_status": "reported",
            "value": true
          }
        },
        "turns": {
          "notes": "Public harness configuration is linked by the creator.",
          "reporting_status": "reported",
          "value": "multi-turn Pi agent"
        }
      },
      "results": [
        {
          "ci_high": 62.68,
          "ci_low": 48.43,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-pi-results-evidence"
          ],
          "metric_id": "accuracy",
          "model_id": "gemini-3-5-flash",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 55.56
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-pi-results-evidence"
          ],
          "metric_id": "mean-cost",
          "model_id": "gemini-3-5-flash",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 1.4254
        },
        {
          "ci_high": null,
          "ci_low": null,
          "confidence": "high",
          "evidence_ids": [
            "spatialbench-repo-pi-results-evidence"
          ],
          "metric_id": "mean-duration",
          "model_id": "gemini-3-5-flash",
          "n": 159,
          "notes": "model_results.csv",
          "status": "verified",
          "value": 1739.29
        }
      ],
      "scope": {
        "filter": null,
        "n": 159,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Exact 159-evaluation Pi row transcribed from the pinned official CSV.",
        "status": "verified"
      },
      "work_id": "spatialbench-repository-release",
      "work_version_id": "spatialbench-repository-release-2026-06-10"
    },
    {
      "benchmark_id": "tape",
      "benchmark_version": "original-2019",
      "comparability_group": "tape-creator-task-native",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "accessed_date": "2026-07-22",
          "id": "tape-creator-full-protocol-evidence",
          "locator": {
            "note": "Defines all five task splits, architectures, training procedures, native metrics, and creator comparison table.",
            "type": "section",
            "value": "Sections 4.2-5 and Table 2; Appendix A"
          },
          "source_id": "tape-paper",
          "source_type": "work",
          "supports": [
            "/scope",
            "/protocol",
            "/metrics"
          ]
        }
      ],
      "id": "tape-creator-full",
      "metrics": [
        {
          "aggregation": "across labeled residues",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "per-residue-accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Per-amino-acid accuracy",
          "tolerance": null,
          "unit": "proportion"
        },
        {
          "aggregation": "per protein then task summary",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "precision-at-l-over-5",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "L/5 medium and long-range precision",
          "tolerance": null,
          "unit": "proportion"
        },
        {
          "aggregation": "held-out test examples",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "classification-accuracy",
          "pass_threshold": null,
          "range": [
            0,
            1
          ],
          "source_label": "Fold-level accuracy",
          "tolerance": null,
          "unit": "proportion"
        },
        {
          "aggregation": "held-out test examples",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "spearman-correlation",
          "pass_threshold": null,
          "range": [
            -1,
            1
          ],
          "source_label": "Spearman's rho",
          "tolerance": null,
          "unit": "correlation"
        }
      ],
      "model_ids": [],
      "protocol": {
        "contamination": {
          "notes": "Split rules differ by task.",
          "reporting_status": "reported",
          "value": "Sequence-identity filtering and biologically motivated held-out splits."
        },
        "grader": {
          "human_review": false,
          "model": null,
          "reporting_status": "reported",
          "type": "deterministic task-specific scorer"
        },
        "reasoning": {
          "notes": "Supervised representation learning; no in-context prompting.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "repeats": {
          "notes": "A common repeat count is not stated for all main-table values.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Exact random seeds are not reported in the main paper.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Supervised representation learning; no in-context prompting.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "statistical": {
          "notes": "Each task retains its native metric.",
          "reporting_status": "reported",
          "value": "Task-level evaluation only; no cross-task normalized aggregate."
        },
        "system_prompt_public": {
          "notes": "Supervised representation learning; no in-context prompting.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "temperature": {
          "notes": "Supervised representation learning; no in-context prompting.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "time_budget": {
          "notes": "No common runtime budget is reported across tasks and architectures.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Supervised representation learning; no in-context prompting.",
          "reporting_status": "not_applicable",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Supervised representation learning; no in-context prompting.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "code_execution": {
            "notes": "Model training is the benchmark procedure rather than an agent tool affordance.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "container": {
            "notes": "The original paper does not prescribe a common container.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Supervised representation learning; no in-context prompting.",
            "reporting_status": "not_applicable",
            "value": null
          },
          "external_tools": {
            "notes": "Alignment-based and one-hot baselines are also evaluated.",
            "reporting_status": "reported",
            "value": "task-specific supervised heads over frozen or fine-tuned protein representations"
          },
          "internet": {
            "notes": "Supervised representation learning; no in-context prompting.",
            "reporting_status": "not_applicable",
            "value": null
          }
        },
        "turns": {
          "notes": "Supervised representation learning; no in-context prompting.",
          "reporting_status": "not_applicable",
          "value": null
        }
      },
      "results": [],
      "scope": {
        "filter": "All five supervised downstream tasks in the creator paper.",
        "n": 5,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Scope and native task metrics are verified; task-specific model result tables are intentionally not flattened into a cross-task ranking.",
        "status": "verified"
      },
      "work_id": "tape-paper",
      "work_version_id": "tape-paper-2019-12-08"
    },
    {
      "benchmark_id": "virbench",
      "benchmark_version": "1.0",
      "comparability_group": "virbench-v1-full-ncbi-virus",
      "entity_type": "evaluation_run",
      "evidence": [
        {
          "locator": "VirBench methods and results",
          "supports": [
            "scope",
            "protocol.tools",
            "protocol.grader",
            "metrics"
          ],
          "work_id": "virbench-official"
        }
      ],
      "id": "virbench-official-run",
      "metrics": [
        {
          "aggregation": "problem-weighted mean",
          "baseline_model_id": null,
          "higher_is_better": true,
          "kind": "absolute",
          "metric_id": "accuracy",
          "pass_threshold": null,
          "range": [
            0,
            100
          ],
          "source_label": "Mean accuracy",
          "tolerance": null,
          "unit": "percent"
        }
      ],
      "model_ids": [],
      "protocol": {
        "contamination": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "grader": {
          "human_review": true,
          "model": null,
          "reporting_status": "reported",
          "type": "manual verified-answer comparison"
        },
        "reasoning": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "repeats": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "seed": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "shots": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "statistical": {
          "notes": "The report gives a system range but not a complete machine-readable result table.",
          "reporting_status": "reported",
          "value": "mean accuracy"
        },
        "system_prompt_public": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "temperature": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "time_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "token_budget": {
          "notes": "Not reported.",
          "reporting_status": "not_reported",
          "value": null
        },
        "tools": {
          "browser": {
            "notes": "Agent uses a retrieval interface.",
            "reporting_status": "reported",
            "value": true
          },
          "code_execution": {
            "notes": "Not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "container": {
            "notes": "Not reported.",
            "reporting_status": "not_reported",
            "value": null
          },
          "databases": {
            "notes": "Primary retrieval source.",
            "reporting_status": "reported",
            "value": [
              "NCBI Virus"
            ]
          },
          "external_tools": {
            "notes": "Exact implementation is not public.",
            "reporting_status": "reported",
            "value": "scientific-agent retrieval tools"
          },
          "internet": {
            "notes": "NCBI Virus is accessed live.",
            "reporting_status": "reported",
            "value": true
          }
        },
        "turns": {
          "notes": "Scientific agents iteratively query and analyze retrieved records.",
          "reporting_status": "reported",
          "value": "multi-turn"
        }
      },
      "results": [],
      "scope": {
        "filter": "120 manually verified questions across 40 pathogens.",
        "n": 120,
        "reporting_status": "reported",
        "selection": null,
        "subset_id": null,
        "type": "full"
      },
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "No model rows are published until exact system labels and values can be source-located.",
        "status": "verified"
      },
      "work_id": "virbench-official",
      "work_version_id": "virbench-official-2025-05-20"
    }
  ],
  "meta": {
    "audit_exemptions": [
      {
        "benchmark_id": "virbench",
        "decision_date": "2026-07-21",
        "reason": "Detailed field-level auditing was intentionally deferred by the maintainer; the verified v1 record remains available with legacy status."
      }
    ],
    "canonical_repository": "https://github.com/SpectrAI-Initiative/bio-benchmark-atlas",
    "canonical_site": "https://spectrai-initiative.github.io/bio-benchmark-atlas/",
    "maintainer": "wang422003",
    "name": "BioBench Atlas",
    "published_source_classes": [
      "benchmark_creator",
      "official_model_provider",
      "independent_reproduction"
    ],
    "release_date": "2026-07-22",
    "version": "1.4.0-dev"
  },
  "models": [
    {
      "aliases": [],
      "entity_type": "model",
      "id": "accelrys-software-inc-discovery-studio",
      "name": "Discovery Studio",
      "provider": "Accelrys Software Inc.",
      "release_date": null,
      "verification": {
        "last_verified": "2026-08-14",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "release 4.0"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "anthropic-claude-mythos-5",
      "name": "Claude Mythos 5",
      "provider": "Anthropic",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-29",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Claude Mythos 5"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "anthropic-claude-opus-5",
      "name": "Claude Opus 5",
      "provider": "Anthropic",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-29",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Claude Opus 5"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "anthropic-claude-sonnet-5",
      "name": "Claude Sonnet 5",
      "provider": "Anthropic",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-29",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Claude Sonnet 5"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "anthropic-opus-4-6",
      "name": "Opus 4.6",
      "provider": "Anthropic",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-30",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Opus 4.6"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "anthropic-opus-4-7",
      "name": "Opus 4.7",
      "provider": "Anthropic",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-30",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Opus 4.7"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "anthropic-opus-4-8",
      "name": "Opus 4.8",
      "provider": "Anthropic",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-30",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Opus 4.8"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "anthropic-sonnet-4-6",
      "name": "Sonnet 4.6",
      "provider": "Anthropic",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-30",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Sonnet 4.6"
    },
    {
      "aliases": [
        "Alpaca-7B"
      ],
      "entity_type": "model",
      "id": "bioinstruction-alpaca-7b",
      "name": "Alpaca-7B (Biology-Instructions label)",
      "provider": "Stanford CRFM",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact upstream checkpoint revision is not reported.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Alpaca-7B"
    },
    {
      "aliases": [
        "BioMedGPT-LM-7B"
      ],
      "entity_type": "model",
      "id": "bioinstruction-biomedgpt-lm-7b",
      "name": "BioMedGPT-LM-7B (Biology-Instructions label)",
      "provider": "BioMedGPT creators",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Reported for RNA, protein, and multi-molecule tasks; no DNA result row is published.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "BioMedGPT-LM-7B"
    },
    {
      "aliases": [
        "ours (stage 1 + balanced stage 2)"
      ],
      "entity_type": "model",
      "id": "bioinstruction-chatmultiomics-balanced",
      "name": "ChatMultiOmics stage 1 + balanced stage 2",
      "provider": "Biology-Instructions creators",
      "release_date": "2024-12-26",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Creator-paper balanced-dataset ablation; a public checkpoint is not linked.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "ours (stage 1 + balanced stage 2)"
    },
    {
      "aliases": [
        "ours (stage 1 + stage 2)"
      ],
      "entity_type": "model",
      "id": "bioinstruction-chatmultiomics-stage12",
      "name": "ChatMultiOmics stage 1 + stage 2",
      "provider": "Biology-Instructions creators",
      "release_date": "2024-12-26",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Creator-paper continued-pretraining plus instruction-tuning system; a public checkpoint is not linked.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "ours (stage 1 + stage 2)"
    },
    {
      "aliases": [
        "ours (stage 1 + stage 2 + stage 3)",
        "ChatMultiOmics"
      ],
      "entity_type": "model",
      "id": "bioinstruction-chatmultiomics-stage123",
      "name": "ChatMultiOmics",
      "provider": "Biology-Instructions creators",
      "release_date": "2024-12-26",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Final three-stage creator system. The official README says its Hugging Face checkpoint will be released but provides no checkpoint link at the audited commit.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "ours (stage 1 + stage 2 + stage 3)"
    },
    {
      "aliases": [
        "ours (stage 2 only)"
      ],
      "entity_type": "model",
      "id": "bioinstruction-chatmultiomics-stage2",
      "name": "ChatMultiOmics stage 2 only",
      "provider": "Biology-Instructions creators",
      "release_date": "2024-12-26",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Creator-paper instruction-tuning-only ablation; a public checkpoint is not linked.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "ours (stage 2 only)"
    },
    {
      "aliases": [
        "Galactica-1.3B"
      ],
      "entity_type": "model",
      "id": "bioinstruction-galactica-13b",
      "name": "Galactica-1.3B (Biology-Instructions label)",
      "provider": "Meta AI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact upstream checkpoint revision is not reported.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Galactica-1.3B"
    },
    {
      "aliases": [
        "GLM-4-9B-Chat",
        "ChatGLM4"
      ],
      "entity_type": "model",
      "id": "bioinstruction-glm4-9b-chat",
      "name": "GLM-4-9B-Chat (Biology-Instructions label)",
      "provider": "Zhipu AI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "The paper uses GLM-4-9B-Chat in text/tables and ChatGLM4 in Figure 5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "GLM-4-9B-Chat"
    },
    {
      "aliases": [
        "GPT-4o"
      ],
      "entity_type": "model",
      "id": "bioinstruction-gpt4o",
      "name": "GPT-4o (Biology-Instructions snapshot not reported)",
      "provider": "OpenAI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "The creator paper reports no dated API snapshot or exact provider version.",
        "status": "verified"
      },
      "version_status": "not_reported",
      "version_string": null
    },
    {
      "aliases": [
        "GPT-4o-mini"
      ],
      "entity_type": "model",
      "id": "bioinstruction-gpt4o-mini",
      "name": "GPT-4o-mini (Biology-Instructions snapshot not reported)",
      "provider": "OpenAI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "The creator paper reports no dated API snapshot or exact provider version.",
        "status": "verified"
      },
      "version_status": "not_reported",
      "version_string": null
    },
    {
      "aliases": [
        "InstructProtein-1.3B"
      ],
      "entity_type": "model",
      "id": "bioinstruction-instructprotein-13b",
      "name": "InstructProtein-1.3B (Biology-Instructions label)",
      "provider": "InstructProtein creators",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact upstream checkpoint revision is not reported.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "InstructProtein-1.3B"
    },
    {
      "aliases": [
        "Llama-molinst-protein-7B (Mol-Ins)",
        "Mol-Ins"
      ],
      "entity_type": "model",
      "id": "bioinstruction-llama-molinst-protein-7b",
      "name": "Llama-molinst-protein-7B (Mol-Ins)",
      "provider": "Mol-Instructions creators",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact upstream checkpoint revision is not reported.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Llama-molinst-protein-7B"
    },
    {
      "aliases": [
        "Llama2-7B-Chat"
      ],
      "entity_type": "model",
      "id": "bioinstruction-llama2-7b-chat",
      "name": "Llama2-7B-Chat (Biology-Instructions label)",
      "provider": "Meta",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Kept separate from similarly named registry models because the paper does not report a weight revision.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Llama2-7B-Chat"
    },
    {
      "aliases": [
        "LLaMA3.1-8B-Instruct"
      ],
      "entity_type": "model",
      "id": "bioinstruction-llama31-8b-instruct",
      "name": "LLaMA3.1-8B-Instruct (Biology-Instructions label)",
      "provider": "Meta",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "The creator paper reports this model label but not a repository revision or weight hash.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "LLaMA3.1-8B-Instruct"
    },
    {
      "aliases": [
        "Qwen2-7B"
      ],
      "entity_type": "model",
      "id": "bioinstruction-qwen2-7b",
      "name": "Qwen2-7B (Biology-Instructions label)",
      "provider": "Qwen",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "The creator paper does not disambiguate base versus instruct checkpoint beyond this label.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Qwen2-7B"
    },
    {
      "aliases": [
        "Vicuna-v1.5-7B",
        "Vicuna"
      ],
      "entity_type": "model",
      "id": "bioinstruction-vicuna15-7b",
      "name": "Vicuna-v1.5-7B (Biology-Instructions label)",
      "provider": "LMSYS Org",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact upstream checkpoint revision is not reported.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Vicuna-v1.5-7B"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "bixbench-claude-3-5-sonnet-20241022",
      "name": "Claude 3.5 Sonnet 20241022",
      "provider": "Anthropic",
      "release_date": "2024-10-22",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact endpoint string in both official v1.5 agent configuration files; not substituted for the unversioned/latest label in other BixBench evaluations.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "anthropic/claude-3-5-sonnet-20241022"
    },
    {
      "aliases": [
        "Claude 3.5 Sonnet",
        "claude-3-5-sonnet-latest"
      ],
      "entity_type": "model",
      "id": "bixbench-claude-3-5-sonnet-unversioned",
      "name": "Claude 3.5 Sonnet (BixBench version not reported)",
      "provider": "Anthropic",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "The evaluation labels do not state the endpoint snapshot. The dated Claude model used during question drafting is not assumed to be the evaluated model.",
        "status": "verified"
      },
      "version_status": "not_reported",
      "version_string": null
    },
    {
      "aliases": [
        "GPT-4o",
        "gpt-4o"
      ],
      "entity_type": "model",
      "id": "bixbench-gpt-4o-unversioned",
      "name": "GPT-4o (BixBench version not reported)",
      "provider": "OpenAI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "The paper and v1.5 result files identify GPT-4o but do not report the evaluated endpoint snapshot or dated model version.",
        "status": "verified"
      },
      "version_status": "not_reported",
      "version_string": null
    },
    {
      "aliases": [
        "Claude 3.5 Sonnet"
      ],
      "entity_type": "model",
      "id": "blade-claude-3-5-sonnet-20240620",
      "name": "Claude 3.5 Sonnet (claude-3-5-sonnet-20240620; BLADE)",
      "provider": "Anthropic",
      "release_date": "2024-06-20",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "The official BLADE model configuration maps its Claude 3.5 Sonnet label to the dated Anthropic model string used by the public baseline scripts.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "claude-3-5-sonnet-20240620"
    },
    {
      "aliases": [
        "CodeLlama 7B",
        "CodeLlama Instruct 7B"
      ],
      "entity_type": "model",
      "id": "blade-codellama-7b-instruct",
      "name": "CodeLlama Instruct 7B (BLADE)",
      "provider": "Meta",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "The creator paper supplies the model family/size and the official BLADE config pins the Hugging Face checkpoint string; no inference-server revision is reported.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "meta-llama/CodeLlama-7b-Instruct-hf"
    },
    {
      "aliases": [
        "Deepseek-Coder 6.7B",
        "DeepSeek-Coder Instruct 6.7B"
      ],
      "entity_type": "model",
      "id": "blade-deepseek-coder-6-7b-instruct",
      "name": "DeepSeek-Coder Instruct 6.7B (BLADE)",
      "provider": "DeepSeek",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "The creator paper supplies the model family/size and the official BLADE config pins the Hugging Face checkpoint string; no inference-server revision is reported.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "deepseek-ai/deepseek-coder-6.7b-instruct"
    },
    {
      "aliases": [
        "Gemini 1.5 Pro",
        "gemini-1.5-pro-latest"
      ],
      "entity_type": "model",
      "id": "blade-gemini-1-5-pro-unversioned",
      "name": "Gemini 1.5 Pro (BLADE snapshot not reported)",
      "provider": "Google",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "The paper reports Gemini 1.5 Pro and the code uses the moving latest alias; no immutable endpoint snapshot is reported.",
        "status": "verified"
      },
      "version_status": "not_reported",
      "version_string": null
    },
    {
      "aliases": [
        "GPT-3.5 Turbo",
        "gpt-3.5-turbo"
      ],
      "entity_type": "model",
      "id": "blade-gpt35-turbo-unversioned",
      "name": "GPT-3.5 Turbo (BLADE snapshot not reported)",
      "provider": "OpenAI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "The paper and official configuration use an undated GPT-3.5 Turbo alias; no dated endpoint snapshot is reported.",
        "status": "verified"
      },
      "version_status": "not_reported",
      "version_string": null
    },
    {
      "aliases": [
        "GPT-4o",
        "gpt-4o"
      ],
      "entity_type": "model",
      "id": "blade-gpt4o-unversioned",
      "name": "GPT-4o (BLADE snapshot not reported)",
      "provider": "OpenAI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "The paper reports GPT-4o for generation and LM-assisted evaluation, while the official Azure configuration leaves the deployment identifier blank; no dated model snapshot is reported.",
        "status": "verified"
      },
      "version_status": "not_reported",
      "version_string": null
    },
    {
      "aliases": [
        "Llama3 70B",
        "Llama 3 70B"
      ],
      "entity_type": "model",
      "id": "blade-llama3-70b-unversioned",
      "name": "Llama 3 70B (BLADE snapshot not reported)",
      "provider": "Meta",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "The paper reports the Llama 3 70B family and size, but the official config contains more than one provider route and the evaluated route/checkpoint revision is not stated.",
        "status": "verified"
      },
      "version_status": "not_reported",
      "version_string": null
    },
    {
      "aliases": [
        "Llama3 8B",
        "Llama 3 8B"
      ],
      "entity_type": "model",
      "id": "blade-llama3-8b-unversioned",
      "name": "Llama 3 8B (BLADE snapshot not reported)",
      "provider": "Meta",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "The paper reports the Llama 3 8B family and size but does not identify a fixed checkpoint revision or unambiguously tie the run to one of the provider aliases in the code config.",
        "status": "verified"
      },
      "version_status": "not_reported",
      "version_string": null
    },
    {
      "aliases": [
        "Mixtral-8x22B",
        "Mixtral 8x22B"
      ],
      "entity_type": "model",
      "id": "blade-mixtral-8x22b-unversioned",
      "name": "Mixtral 8x22B (BLADE snapshot not reported)",
      "provider": "Mistral AI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "The paper reports the family and size, but the official config contains Mistral and Together routes and does not identify which exact served snapshot produced the published results.",
        "status": "verified"
      },
      "version_status": "not_reported",
      "version_string": null
    },
    {
      "aliases": [
        "AF3 CAMEO baseline"
      ],
      "entity_type": "model",
      "id": "cameo-alphafold3-v301",
      "name": "AlphaFold 3 v3.0.1 (CAMEO baseline)",
      "provider": "Google DeepMind and Isomorphic Labs",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact baseline version reported in the final CAMEO creator paper.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "3.0.1"
    },
    {
      "aliases": [
        "MultiFOLD"
      ],
      "entity_type": "model",
      "id": "cameo-multifold-2024",
      "name": "MultiFOLD (CAMEO 2024 participant)",
      "provider": "University of Reading",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "The final CAMEO creator paper identifies MultiFOLD but does not provide a release string for the 2024 server.",
        "status": "verified"
      },
      "version_status": "not_reported",
      "version_string": null
    },
    {
      "aliases": [
        "SWISS-MODEL"
      ],
      "entity_type": "model",
      "id": "cameo-swissmodel-2024",
      "name": "SWISS-MODEL (CAMEO 2024 participant)",
      "provider": "SIB Swiss Institute of Bioinformatics",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "The final CAMEO creator paper identifies the participant but does not provide an exact server release string.",
        "status": "verified"
      },
      "version_status": "not_reported",
      "version_string": null
    },
    {
      "aliases": [
        "SM_Glide"
      ],
      "entity_type": "model",
      "id": "cameo-swissmodel-glide",
      "name": "SWISS-MODEL + Schrödinger Glide",
      "provider": "CAMEO and SWISS-MODEL",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact baseline label is reported; the creator paper does not report an exact Glide release string.",
        "status": "verified"
      },
      "version_status": "not_reported",
      "version_string": null
    },
    {
      "aliases": [
        "SM_Vina_ad4"
      ],
      "entity_type": "model",
      "id": "cameo-swissmodel-vina-ad4",
      "name": "SWISS-MODEL + AutoDock Vina (AutoDock4 scoring)",
      "provider": "CAMEO and SWISS-MODEL",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Creator-paper baseline; receptor preparation and ligand tool versions are documented in Methods 3.2.1.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "AutoDock Vina 1.2.5; AutoDock4 scoring function"
    },
    {
      "aliases": [
        "SM_Vina_vina"
      ],
      "entity_type": "model",
      "id": "cameo-swissmodel-vina-vina",
      "name": "SWISS-MODEL + AutoDock Vina (vina scoring)",
      "provider": "CAMEO and SWISS-MODEL",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Creator-paper baseline; receptor preparation and ligand tool versions are documented in Methods 3.2.1.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "AutoDock Vina 1.2.5; vina scoring function"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "carp-38m",
      "name": "CARP (38M)",
      "provider": "Microsoft Research",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact size variant in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "CARP 38M"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "carp-600k",
      "name": "CARP (600K)",
      "provider": "Microsoft Research",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact size variant in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "CARP 600K"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "carp-640m",
      "name": "CARP (640M)",
      "provider": "Microsoft Research",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact size variant in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "CARP 640M"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "carp-76m",
      "name": "CARP (76M)",
      "provider": "Microsoft Research",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact size variant in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "CARP 76M"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "chatgpt-5-2",
      "name": "ChatGPT 5.2",
      "provider": "OpenAI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact non-agentic API baseline label in CompBioBench; endpoint snapshot and release date are not reported.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "ChatGPT 5.2"
    },
    {
      "aliases": [
        "Claude Code (Haiku 4.5)"
      ],
      "entity_type": "model",
      "id": "claude-code-haiku-4-5",
      "name": "Claude Code (Haiku 4.5)",
      "provider": "Anthropic",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact CLI version and dated model identifier reported in the CompBioBench Methods; effort was unsupported.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Claude Code v2.1.87 + claude-haiku-4-5-20251001"
    },
    {
      "aliases": [
        "Claude Code (Opus 4.6) [max]"
      ],
      "entity_type": "model",
      "id": "claude-code-opus-4-6",
      "name": "Claude Code (Opus 4.6)",
      "provider": "Anthropic",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact CLI version, model identifier, context variant, and max effort reported in the CompBioBench Methods.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Claude Code v2.1.87 + claude-opus-4-6 (1M context)"
    },
    {
      "aliases": [
        "Claude Code (Sonnet 4.6) [high]"
      ],
      "entity_type": "model",
      "id": "claude-code-sonnet-4-6",
      "name": "Claude Code (Sonnet 4.6)",
      "provider": "Anthropic",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact CLI version, model identifier, context variant, and high effort reported in the CompBioBench Methods.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Claude Code v2.1.87 + claude-sonnet-4-6 (1M context)"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "claude-haiku-4-5",
      "name": "Claude Haiku 4.5",
      "provider": "Anthropic",
      "release_date": "2025-10-15",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact BioMysteryBench Figure 1–3 label; release date verified from Anthropic's official model announcement.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Claude Haiku 4.5"
    },
    {
      "aliases": [
        "Mythos Preview"
      ],
      "entity_type": "model",
      "id": "claude-mythos-preview",
      "name": "Claude Mythos Preview",
      "provider": "Anthropic",
      "release_date": "2026-04-07",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact BioMysteryBench Figure 1–3 label; the limited-access preview was officially announced with Project Glasswing on 2026-04-07.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Claude Mythos Preview"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "claude-opus-4",
      "name": "Claude Opus 4",
      "provider": "Anthropic",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Identity follows the exact label or API string reported by the cited official work; no unreported dated snapshot is inferred.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Claude Opus 4"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "claude-opus-4-1",
      "name": "Claude Opus 4.1",
      "provider": "Anthropic",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Identity follows the exact label or API string reported by the cited official work; no unreported dated snapshot is inferred.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Claude Opus 4.1"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "claude-opus-4-5",
      "name": "Claude Opus 4.5",
      "provider": "Anthropic",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Exact label in the SpatialBench paper, repository, and Anthropic official healthcare/life-sciences page; no dated endpoint snapshot is inferred.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Claude Opus 4.5"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "claude-opus-4-6",
      "name": "Claude Opus 4.6",
      "provider": "Anthropic",
      "release_date": "2026-02-05",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact system-card and BioMysteryBench Figure 1–3 label.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Claude Opus 4.6"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "claude-opus-4-7",
      "name": "Claude Opus 4.7",
      "provider": "Anthropic",
      "release_date": "2026-04-16",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact BioMysteryBench Figure 1–3 label; release date verified from Anthropic's official model announcement.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Claude Opus 4.7"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "claude-opus-4-8",
      "name": "Claude Opus 4.8",
      "provider": "Anthropic",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact GeneBench-Pro Supplementary Table 1 model label; endpoint snapshot and release date are not reported in the paper.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Claude Opus 4.8"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "claude-sonnet-4",
      "name": "Claude Sonnet 4",
      "provider": "Anthropic",
      "release_date": "2025-05-22",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Official release label; dated API snapshot not specified in the cited evaluation.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Claude Sonnet 4"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "claude-sonnet-4-5",
      "name": "Claude Sonnet 4.5",
      "provider": "Anthropic",
      "release_date": "2025-09-29",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Official release label; dated API snapshot not specified in the cited evaluation.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Claude Sonnet 4.5"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "claude-sonnet-4-6",
      "name": "Claude Sonnet 4.6",
      "provider": "Anthropic",
      "release_date": "2026-02-17",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact system-card and BioMysteryBench Figure 1–3 label.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Claude Sonnet 4.6"
    },
    {
      "aliases": [
        "Codex CLI (GPT 5.4) [xhigh]"
      ],
      "entity_type": "model",
      "id": "codex-cli-gpt-5-4",
      "name": "Codex CLI (GPT-5.4)",
      "provider": "OpenAI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact CLI version, model identifier, and xhigh reasoning setting reported in the CompBioBench Methods.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Codex CLI v0.115.0 + gpt-5.4"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "deepseek-v4-flash",
      "name": "DeepSeek V4 Flash",
      "provider": "DeepSeek",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact GeneBench-Pro Supplementary Table 1 model label; endpoint snapshot and release date are not reported in the paper.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "DeepSeek V4 Flash"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "deepseek-v4-pro",
      "name": "DeepSeek V4 Pro",
      "provider": "DeepSeek",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact GeneBench-Pro Supplementary Table 1 model label; endpoint snapshot and release date are not reported in the paper.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "DeepSeek V4 Pro"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "deepsequence-ensemble",
      "name": "DeepSequence (ensemble)",
      "provider": "Harvard University",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact result variant in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "DeepSequence ensemble"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "deepsequence-single",
      "name": "DeepSequence (single)",
      "provider": "Harvard University",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact result variant in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "DeepSequence single"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "esm-1b",
      "name": "ESM-1b",
      "provider": "Meta AI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact label in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "ESM-1b"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "esm-1v-ensemble",
      "name": "ESM-1v (ensemble)",
      "provider": "Meta AI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact result variant in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "ESM-1v ensemble"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "esm-1v-single",
      "name": "ESM-1v (single)",
      "provider": "Meta AI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact result variant in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "ESM-1v single"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "esm-if1",
      "name": "ESM-IF1",
      "provider": "Meta AI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact label in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "ESM-IF1"
    },
    {
      "aliases": [
        "ESM-2 150M"
      ],
      "entity_type": "model",
      "id": "esm2-150m",
      "name": "ESM2 (150M)",
      "provider": "Meta AI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact size variant in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "ESM2 150M"
    },
    {
      "aliases": [
        "ESM-2 15B"
      ],
      "entity_type": "model",
      "id": "esm2-15b",
      "name": "ESM2 (15B)",
      "provider": "Meta AI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact size variant in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "ESM2 15B"
    },
    {
      "aliases": [
        "ESM-2 35M"
      ],
      "entity_type": "model",
      "id": "esm2-35m",
      "name": "ESM2 (35M)",
      "provider": "Meta AI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact size variant in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "ESM2 35M"
    },
    {
      "aliases": [
        "ESM-2 3B"
      ],
      "entity_type": "model",
      "id": "esm2-3b",
      "name": "ESM2 (3B)",
      "provider": "Meta AI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact size variant in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "ESM2 3B"
    },
    {
      "aliases": [
        "ESM-2 650M"
      ],
      "entity_type": "model",
      "id": "esm2-650m",
      "name": "ESM2 (650M)",
      "provider": "Meta AI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact size variant in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "ESM2 650M"
    },
    {
      "aliases": [
        "ESM-2 8M"
      ],
      "entity_type": "model",
      "id": "esm2-8m",
      "name": "ESM2 (8M)",
      "provider": "Meta AI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact size variant in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "ESM2 8M"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "eve-ensemble",
      "name": "EVE (ensemble)",
      "provider": "Harvard Medical School",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact result variant in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "EVE ensemble"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "eve-single",
      "name": "EVE (single)",
      "provider": "Harvard Medical School",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact result variant in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "EVE single"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "evmutation",
      "name": "EVmutation",
      "provider": "Harvard Medical School",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact label in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "EVmutation"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "flip-blosum62",
      "name": "BLOSUM62 baseline (FLIP)",
      "provider": "FLIP creators",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Parameter-free creator baseline; not applicable to AAV indels or diverse Meltome sequences.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "BLOSUM62 score relative to wild type"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "flip-cnn",
      "name": "Convolutional network (FLIP)",
      "provider": "FLIP creators",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "One-hot CNN with width-5/1024-channel convolution, 2048-dimensional mapping, max pooling, and scalar output.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "FLIP CNN"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "flip-esm-untrained-mean",
      "name": "ESM-untrained (mean)",
      "provider": "FLIP creators",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Randomly initialized 750M ESM architecture, mean pooled, with the FLIP supervised predictor.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "ESM-untrained (mean)"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "flip-esm-untrained-mut-mean",
      "name": "ESM-untrained (mut mean)",
      "provider": "FLIP creators",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Randomly initialized 750M ESM architecture pooled over the mutated region; not applicable to Meltome.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "ESM-untrained (mut mean)"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "flip-esm-untrained-per-aa",
      "name": "ESM-untrained (per AA)",
      "provider": "FLIP creators",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Randomly initialized 750M ESM architecture with learned 1D-attention pooling/regression head.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "ESM-untrained (per AA)"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "flip-esm1b-mean",
      "name": "ESM-1b (mean; FLIP head)",
      "provider": "Meta AI / FLIP creators",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Frozen 750M ESM-1b embeddings, mean pooled, with the FLIP supervised predictor.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "ESM-1b (mean)"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "flip-esm1b-mut-mean",
      "name": "ESM-1b (mut mean; FLIP head)",
      "provider": "Meta AI / FLIP creators",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Frozen 750M ESM-1b embeddings mean-pooled over the mutated region; not applicable to Meltome.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "ESM-1b (mut mean)"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "flip-esm1b-per-aa",
      "name": "ESM-1b (per AA; FLIP head)",
      "provider": "Meta AI / FLIP creators",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Frozen 750M ESM-1b embeddings with a learned 1D-attention pooling/regression head; exact creator-paper label.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "ESM-1b (per AA)"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "flip-esm1v-mean",
      "name": "ESM-1v (mean; FLIP head)",
      "provider": "Meta AI / FLIP creators",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "One 750M ESM-1v ensemble member, mean pooled, with the FLIP supervised predictor.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "ESM-1v (mean; one ensemble member)"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "flip-esm1v-mut-mean",
      "name": "ESM-1v (mut mean; FLIP head)",
      "provider": "Meta AI / FLIP creators",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "One 750M ESM-1v ensemble member mean-pooled over the mutated region; not applicable to Meltome.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "ESM-1v (mut mean; one ensemble member)"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "flip-esm1v-per-aa",
      "name": "ESM-1v (per AA; FLIP head)",
      "provider": "Meta AI / FLIP creators",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "One 750M ESM-1v ensemble member with learned 1D-attention pooling/regression head, as reported due to compute constraints.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "ESM-1v (per AA; one ensemble member)"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "flip-levenshtein",
      "name": "Levenshtein distance baseline (FLIP)",
      "provider": "FLIP creators",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Parameter-free creator baseline; not applicable to the diverse Meltome sequences.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Levenshtein distance to wild type"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "flip-ridge",
      "name": "Ridge regression (FLIP)",
      "provider": "FLIP creators",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact supervised baseline family and settings reported in Section 4.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "scikit-learn Ridge defaults on one-hot encoding"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "gemini-2-5-pro",
      "name": "Gemini 2.5 Pro",
      "provider": "Google",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Exact SpatialBench source label; no dated endpoint snapshot is reported.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Gemini 2.5 Pro"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "gemini-3-1-pro",
      "name": "Gemini 3.1 Pro",
      "provider": "Google",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact label in both LifeSciBench and GeneBench-Pro; neither report states an endpoint snapshot, and GeneBench-Pro does not report a release date.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Gemini 3.1 Pro"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "gemini-3-1-pro-preview",
      "name": "gemini-3.1-pro-preview",
      "provider": "Google",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Exact model_name in the pinned SpatialBench result CSV.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "gemini-3.1-pro-preview"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "gemini-3-5-flash",
      "name": "Gemini 3.5 Flash",
      "provider": "Google",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact GeneBench-Pro Supplementary Table 1 model label; endpoint snapshot and release date are not reported in the paper.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Gemini 3.5 Flash"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "gemme",
      "name": "GEMME",
      "provider": "Sorbonne Université",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact label in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "GEMME"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "glm-5-1",
      "name": "GLM 5.1",
      "provider": "Zhipu AI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact GeneBench-Pro Supplementary Table 1 model label; endpoint snapshot and release date are not reported in the paper.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "GLM 5.1"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "glm-5-2",
      "name": "GLM 5.2",
      "provider": "Zhipu AI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact GeneBench-Pro Supplementary Table 1 model label; endpoint snapshot and release date are not reported in the paper.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "GLM 5.2"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "gpt-5-1",
      "name": "GPT-5.1",
      "provider": "OpenAI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Exact SpatialBench source label; no dated endpoint snapshot is reported.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "GPT-5.1"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "gpt-5-2",
      "name": "GPT-5.2",
      "provider": "OpenAI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact GeneBench-Pro Supplementary Table 1 model label; endpoint snapshot and release date are not reported in the paper.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "GPT-5.2"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "gpt-5-2-pro",
      "name": "GPT-5.2 Pro (Extended)",
      "provider": "OpenAI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact GeneBench-Pro Supplementary Table 1 model label; endpoint snapshot and release date are not reported in the paper.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "GPT-5.2 Pro (Extended)"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "gpt-5-4",
      "name": "GPT-5.4",
      "provider": "OpenAI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact public-report and GeneBench-Pro Supplementary Table 1 label; the evaluated endpoint snapshot and release date are not reported.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "GPT-5.4"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "gpt-5-4-pro",
      "name": "GPT-5.4 Pro (Extended)",
      "provider": "OpenAI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact GeneBench-Pro Supplementary Table 1 model label; endpoint snapshot and release date are not reported in the paper.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "GPT-5.4 Pro (Extended)"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "gpt-5-5",
      "name": "GPT-5.5",
      "provider": "OpenAI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact public-report and GeneBench-Pro Supplementary Table 1 label; the evaluated endpoint snapshot and release date are not reported.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "GPT-5.5"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "gpt-5-5-pro",
      "name": "GPT-5.5 Pro (Extended)",
      "provider": "OpenAI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact GeneBench-Pro Supplementary Table 1 model label; endpoint snapshot and release date are not reported in the paper.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "GPT-5.5 Pro (Extended)"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "gpt-5-6-luna",
      "name": "GPT-5.6 Luna",
      "provider": "OpenAI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact GeneBench-Pro Supplementary Table 1 model label; endpoint snapshot and release date are not reported in the paper.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "GPT-5.6 Luna"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "gpt-5-6-luna-pro",
      "name": "GPT-5.6 Luna Pro (Extended)",
      "provider": "OpenAI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact GeneBench-Pro Supplementary Table 1 model label; endpoint snapshot and release date are not reported in the paper.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "GPT-5.6 Luna Pro (Extended)"
    },
    {
      "aliases": [
        "GPT-5.6 Sol (Pro)",
        "GPT-5.6 Pro mode"
      ],
      "entity_type": "model",
      "id": "gpt-5-6-pro",
      "name": "GPT-5.6 Sol Pro (Extended)",
      "provider": "OpenAI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "GeneBench-Pro reports the exact label GPT-5.6 Sol Pro (Extended); earlier registry sources used GPT-5.6 Sol (Pro) or GPT-5.6 Pro mode. The endpoint snapshot and release date are not reported.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "GPT-5.6 Sol Pro (Extended)"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "gpt-5-6-sol",
      "name": "GPT-5.6 Sol",
      "provider": "OpenAI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact label in LifeSciBench and GeneBench-Pro; the reports do not state the evaluated API endpoint snapshot, and GeneBench-Pro does not report a release date.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "GPT-5.6 Sol"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "gpt-5-6-terra",
      "name": "GPT-5.6 Terra",
      "provider": "OpenAI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact GeneBench-Pro Supplementary Table 1 model label; endpoint snapshot and release date are not reported in the paper.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "GPT-5.6 Terra"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "gpt-5-6-terra-pro",
      "name": "GPT-5.6 Terra Pro (Extended)",
      "provider": "OpenAI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact GeneBench-Pro Supplementary Table 1 model label; endpoint snapshot and release date are not reported in the paper.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "GPT-5.6 Terra Pro (Extended)"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "gpt-rosalind",
      "name": "GPT-Rosalind",
      "provider": "OpenAI",
      "release_date": "2026-06-17",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact label used in the LifeSciBench report.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "GPT-Rosalind"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "grok-4",
      "name": "Grok-4",
      "provider": "xAI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Exact SpatialBench source label; no dated endpoint snapshot is reported.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Grok-4"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "grok-4-1-fast-reasoning",
      "name": "grok-4-1-fast-reasoning",
      "provider": "xAI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Exact model_name in the pinned SpatialBench result CSV; not merged with the paper's under-specified Grok-4.1 label.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "grok-4-1-fast-reasoning"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "grok-4-20-beta-0309-reasoning",
      "name": "grok-4.20-beta-0309-reasoning",
      "provider": "xAI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Exact model_name in the pinned SpatialBench result CSV.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "grok-4.20-beta-0309-reasoning"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "grok-4-3",
      "name": "Grok 4.3",
      "provider": "xAI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact label in both LifeSciBench and GeneBench-Pro; neither report states an endpoint snapshot, and GeneBench-Pro does not report a release date.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Grok 4.3"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "hu-et-al-supervised-contextpred-sup-cp",
      "name": "supervised_contextpred (Sup-CP)",
      "provider": "Hu et al.",
      "release_date": null,
      "verification": {
        "last_verified": "2026-08-02",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "supervised_contextpred"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "hu-et-al-supervised-sup",
      "name": "supervised (Sup)",
      "provider": "Hu et al.",
      "release_date": null,
      "verification": {
        "last_verified": "2026-08-02",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "supervised"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "jumper-et-al-alphafold2",
      "name": "AlphaFold2",
      "provider": "Jumper et al.",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-31",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "model_1_ptm; model_1_multimer_v3 for multichain states"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "kim-zhou-and-chen-hippo",
      "name": "HIPPO",
      "provider": "Kim, Zhou, and Chen",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-31",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "not_reported",
      "version_string": null
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "kimi-k2-6",
      "name": "Kimi K2.6",
      "provider": "Moonshot AI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact GeneBench-Pro Supplementary Table 1 model label; endpoint snapshot and release date are not reported in the paper.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Kimi K2.6"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "kimi-k2-7-code",
      "name": "Kimi K2.7 Code",
      "provider": "Moonshot AI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact GeneBench-Pro Supplementary Table 1 model label; endpoint snapshot and release date are not reported in the paper.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Kimi K2.7 Code"
    },
    {
      "aliases": [
        "Claude 3.5 Sonnet"
      ],
      "entity_type": "model",
      "id": "lab-bench-claude-3-5-sonnet-20240620",
      "name": "Claude 3.5 Sonnet (claude-3-5-sonnet-20240620)",
      "provider": "Anthropic",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Identity follows the exact label or API string reported by the cited official work; no unreported dated snapshot is inferred.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "claude-3-5-sonnet-20240620"
    },
    {
      "aliases": [
        "Claude 3 Haiku"
      ],
      "entity_type": "model",
      "id": "lab-bench-claude-3-haiku-20240307",
      "name": "Claude 3 Haiku (claude-3-haiku-20240307)",
      "provider": "Anthropic",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Identity follows the exact label or API string reported by the cited official work; no unreported dated snapshot is inferred.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "claude-3-haiku-20240307"
    },
    {
      "aliases": [
        "Claude 3 Opus"
      ],
      "entity_type": "model",
      "id": "lab-bench-claude-3-opus-20240229",
      "name": "Claude 3 Opus (claude-3-opus-20240229)",
      "provider": "Anthropic",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Identity follows the exact label or API string reported by the cited official work; no unreported dated snapshot is inferred.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "claude-3-opus-20240229"
    },
    {
      "aliases": [
        "Gemini 1.5 Pro"
      ],
      "entity_type": "model",
      "id": "lab-bench-gemini-1-5-pro-001",
      "name": "Gemini 1.5 Pro (gemini-1.5-pro-001)",
      "provider": "Google",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Identity follows the exact label or API string reported by the cited official work; no unreported dated snapshot is inferred.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "gemini-1.5-pro-001"
    },
    {
      "aliases": [
        "GPT-4 Turbo",
        "gpt-4-turbo"
      ],
      "entity_type": "model",
      "id": "lab-bench-gpt-4-turbo-unversioned",
      "name": "GPT-4 Turbo (LAB-Bench snapshot not reported)",
      "provider": "OpenAI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Identity follows the exact label or API string reported by the cited official work; no unreported dated snapshot is inferred.",
        "status": "verified"
      },
      "version_status": "not_reported",
      "version_string": null
    },
    {
      "aliases": [
        "GPT-4o",
        "gpt-4o"
      ],
      "entity_type": "model",
      "id": "lab-bench-gpt-4o-unversioned",
      "name": "GPT-4o (LAB-Bench snapshot not reported)",
      "provider": "OpenAI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Identity follows the exact label or API string reported by the cited official work; no unreported dated snapshot is inferred.",
        "status": "verified"
      },
      "version_status": "not_reported",
      "version_string": null
    },
    {
      "aliases": [
        "Llama 3 70B Instruct"
      ],
      "entity_type": "model",
      "id": "lab-bench-meta-llama-3-70b-instruct",
      "name": "Meta-Llama-3-70B-Instruct (Anyscale API)",
      "provider": "Meta",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Identity follows the exact label or API string reported by the cited official work; no unreported dated snapshot is inferred.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Meta-Llama-3-70B-Instruct"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "liu-et-al-benchmark-algorithm",
      "name": "Benchmark algorithm",
      "provider": "Liu et al.",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-29",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "not_reported",
      "version_string": null
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "liu-et-al-pp-anb",
      "name": "PP.ANB",
      "provider": "Liu et al.",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-31",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "not_reported",
      "version_string": null
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "liu-et-al-qq-anb",
      "name": "QQ.ANB",
      "provider": "Liu et al.",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-31",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "not_reported",
      "version_string": null
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "liu-et-al-wdist-med",
      "name": "Wdist.med",
      "provider": "Liu et al.",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-31",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "not_reported",
      "version_string": null
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "meier-et-al-esm-1v",
      "name": "ESM-1v",
      "provider": "Meier et al.",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-31",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "esm1v_t33_650M_UR90S_1"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "mif",
      "name": "MIF",
      "provider": "Microsoft Research",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact label in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "MIF"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "mif-st",
      "name": "MIF-ST",
      "provider": "Microsoft Research",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact label in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "MIF-ST"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "mimo-v2-5",
      "name": "MiMo V2.5",
      "provider": "Xiaomi",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact GeneBench-Pro Supplementary Table 1 model label; endpoint snapshot and release date are not reported in the paper.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "MiMo V2.5"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "mimo-v2-5-pro",
      "name": "MiMo V2.5 Pro",
      "provider": "Xiaomi",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact GeneBench-Pro Supplementary Table 1 model label; endpoint snapshot and release date are not reported in the paper.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "MiMo V2.5 Pro"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "minimax-m2-7",
      "name": "MiniMax M2.7",
      "provider": "MiniMax",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact GeneBench-Pro Supplementary Table 1 model label; endpoint snapshot and release date are not reported in the paper.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "MiniMax M2.7"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "minimax-m3",
      "name": "MiniMax M3",
      "provider": "MiniMax",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact GeneBench-Pro Supplementary Table 1 model label; endpoint snapshot and release date are not reported in the paper.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "MiniMax M3"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "msa-transformer-ensemble",
      "name": "MSA Transformer (ensemble)",
      "provider": "Meta AI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact result variant in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "MSA Transformer ensemble"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "msa-transformer-single",
      "name": "MSA Transformer (single)",
      "provider": "Meta AI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact result variant in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "MSA Transformer single"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "not-reported-antiberty",
      "name": "AntiBERTy",
      "provider": "Not reported",
      "release_date": null,
      "verification": {
        "last_verified": "2026-08-20",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "AntiBERTy"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "not-reported-antifold",
      "name": "AntiFold",
      "provider": "Not reported",
      "release_date": null,
      "verification": {
        "last_verified": "2026-08-20",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "AntiFold"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "not-reported-balm-ppi",
      "name": "BALM-PPI",
      "provider": "Not reported",
      "release_date": null,
      "verification": {
        "last_verified": "2026-08-20",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "not_reported",
      "version_string": null
    },
    {
      "aliases": [
        "Baseline"
      ],
      "entity_type": "model",
      "id": "not-reported-balm-ppi-standard-regression-baseline",
      "name": "BALM-PPI standard regression baseline",
      "provider": "Not reported",
      "release_date": null,
      "verification": {
        "last_verified": "2026-08-20",
        "notes": "Exact model identity accepted by the automated double-pass paper review and clarified from the paper's architectural description; production inclusion remains pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "not_reported",
      "version_string": null
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "not-reported-balm-ppi-without-peft",
      "name": "BALM-PPI without PEFT",
      "provider": "Not reported",
      "release_date": null,
      "verification": {
        "last_verified": "2026-08-20",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "not_reported",
      "version_string": null
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "not-reported-currab",
      "name": "CurrAb",
      "provider": "Not reported",
      "release_date": null,
      "verification": {
        "last_verified": "2026-08-20",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "CurrAb"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "not-reported-ddfire",
      "name": "dDFIRE",
      "provider": "Not reported",
      "release_date": null,
      "verification": {
        "last_verified": "2026-08-14",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "not_reported",
      "version_string": null
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "not-reported-dfire",
      "name": "DFIRE",
      "provider": "Not reported",
      "release_date": null,
      "verification": {
        "last_verified": "2026-08-14",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "not_reported",
      "version_string": null
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "not-reported-diffab",
      "name": "DiffAb",
      "provider": "Not reported",
      "release_date": null,
      "verification": {
        "last_verified": "2026-08-20",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "DiffAb"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "not-reported-diffab-fixbb",
      "name": "DiffAb fixbb",
      "provider": "Not reported",
      "release_date": null,
      "verification": {
        "last_verified": "2026-08-20",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "DiffAb fixbb"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "not-reported-dymean",
      "name": "dyMEAN",
      "provider": "Not reported",
      "release_date": null,
      "verification": {
        "last_verified": "2026-08-20",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "dyMEAN"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "not-reported-dymean-fixbb",
      "name": "dyMEAN fixbb",
      "provider": "Not reported",
      "release_date": null,
      "verification": {
        "last_verified": "2026-08-20",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "dyMEAN fixbb"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "not-reported-esm2",
      "name": "ESM2",
      "provider": "Not reported",
      "release_date": null,
      "verification": {
        "last_verified": "2026-08-20",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "ESM2"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "not-reported-esm3",
      "name": "ESM3",
      "provider": "Not reported",
      "release_date": null,
      "verification": {
        "last_verified": "2026-08-20",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "ESM3"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "not-reported-foldx",
      "name": "FoldX",
      "provider": "Not reported",
      "release_date": null,
      "verification": {
        "last_verified": "2026-08-14",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "release 3.0"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "not-reported-mean",
      "name": "MEAN",
      "provider": "Not reported",
      "release_date": null,
      "verification": {
        "last_verified": "2026-08-20",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "MEAN"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "not-reported-mean-fixbb",
      "name": "MEAN fixbb",
      "provider": "Not reported",
      "release_date": null,
      "verification": {
        "last_verified": "2026-08-20",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "MEAN fixbb"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "not-reported-progen2-large",
      "name": "ProGen2-large",
      "provider": "Not reported",
      "release_date": null,
      "verification": {
        "last_verified": "2026-08-20",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "ProGen2-large"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "not-reported-prosst",
      "name": "ProSST",
      "provider": "Not reported",
      "release_date": null,
      "verification": {
        "last_verified": "2026-08-20",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "ProSST"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "not-reported-rosetta",
      "name": "Rosetta",
      "provider": "Not reported",
      "release_date": null,
      "verification": {
        "last_verified": "2026-08-14",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "release 3.1"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "not-reported-saprot",
      "name": "SaProt",
      "provider": "Not reported",
      "release_date": null,
      "verification": {
        "last_verified": "2026-08-20",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "SaProt"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "not-reported-statium",
      "name": "STATIUM",
      "provider": "Not reported",
      "release_date": null,
      "verification": {
        "last_verified": "2026-08-14",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "not_reported",
      "version_string": null
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "openai-gpt-4o-2024-05-13",
      "name": "gpt-4o-2024-05-13",
      "provider": "OpenAI",
      "release_date": "2024-05-13",
      "verification": {
        "last_verified": "2026-08-13",
        "notes": "Exact API snapshot string in the pinned official SOAR configuration.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "gpt-4o-2024-05-13"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "openai-gpt-4o-mini-2024-07-18",
      "name": "gpt-4o-mini-2024-07-18",
      "provider": "OpenAI",
      "release_date": "2024-07-18",
      "verification": {
        "last_verified": "2026-08-13",
        "notes": "Exact API snapshot string in the pinned official SOAR configuration.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "gpt-4o-mini-2024-07-18"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "progen2-base",
      "name": "ProGen2 Base",
      "provider": "Salesforce Research",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact size variant in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "ProGen2 Base"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "progen2-l",
      "name": "ProGen2 L",
      "provider": "Salesforce Research",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact size variant in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "ProGen2 L"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "progen2-m",
      "name": "ProGen2 M",
      "provider": "Salesforce Research",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact size variant in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "ProGen2 M"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "progen2-s",
      "name": "ProGen2 S",
      "provider": "Salesforce Research",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact size variant in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "ProGen2 S"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "progen2-xl",
      "name": "ProGen2 XL",
      "provider": "Salesforce Research",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact size variant in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "ProGen2 XL"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "proteinlmbench-baichuan2-7b",
      "name": "Baichuan2-7B",
      "provider": "Baichuan Intelligence",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact creator-paper label.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Baichuan2-7B"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "proteinlmbench-chatglm3-6b",
      "name": "ChatGLM3-6B",
      "provider": "Zhipu AI / Tsinghua University",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact creator-paper label.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "ChatGLM3-6B"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "proteinlmbench-falcon-7b",
      "name": "Falcon-7B",
      "provider": "Technology Innovation Institute",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact creator-paper Table 3 label.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Falcon-7B"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "proteinlmbench-falcon-7b-instruct",
      "name": "Falcon-7B-Instruct",
      "provider": "Technology Innovation Institute",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact creator-paper Table 3 label.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Falcon-7B-Instruct"
    },
    {
      "aliases": [
        "GPT3.5-turbo"
      ],
      "entity_type": "model",
      "id": "proteinlmbench-gpt35-turbo",
      "name": "GPT3.5-turbo (ProteinLMBench label)",
      "provider": "OpenAI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "The paper does not report the dated API snapshot.",
        "status": "verified"
      },
      "version_status": "not_reported",
      "version_string": null
    },
    {
      "aliases": [
        "GPT4.0-turbo"
      ],
      "entity_type": "model",
      "id": "proteinlmbench-gpt4-turbo",
      "name": "GPT4.0-turbo (ProteinLMBench label)",
      "provider": "OpenAI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "The paper does not report the dated API snapshot.",
        "status": "verified"
      },
      "version_status": "not_reported",
      "version_string": null
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "proteinlmbench-internlm-chat-20b",
      "name": "InternLM-Chat-20B",
      "provider": "Shanghai AI Laboratory",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact creator-paper label.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "InternLM-Chat-20B"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "proteinlmbench-internlm2-20b",
      "name": "InternLM2-20B",
      "provider": "Shanghai AI Laboratory",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact creator-paper Table 3 label; checkpoint revision is not reported.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "InternLM2-20B"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "proteinlmbench-internlm2-7b",
      "name": "InternLM2-7B",
      "provider": "Shanghai AI Laboratory",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Base model in the no-training condition.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "InternLM2-7B"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "proteinlmbench-internlm2-chat-20b",
      "name": "InternLM2-Chat-20B",
      "provider": "Shanghai AI Laboratory",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact creator-paper label.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "InternLM2-Chat-20B"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "proteinlmbench-internlm2-chat-7b",
      "name": "InternLM2-Chat-7B",
      "provider": "Shanghai AI Laboratory",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact creator-paper label.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "InternLM2-Chat-7B"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "proteinlmbench-internlm2-protein-7b-no-ssl",
      "name": "InternLM2-Protein-7B (w/o SSL)",
      "provider": "Toursun Synbio",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "InternLM2-7B trained only with the ProteinLMDataset SFT phase.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "InternLM2-Protein-7B (w/o SSL)"
    },
    {
      "aliases": [
        "Llama2-7B"
      ],
      "entity_type": "model",
      "id": "proteinlmbench-llama2-7b-chat",
      "name": "Llama-2-7B-Chat-hf",
      "provider": "Meta AI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Section 6 reports Llama-2-7B-Chat-hf; Table 3 uses the shortened label Llama2-7B.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Llama-2-7B-Chat-hf"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "proteinlmbench-mistral-7b-instruct-v02",
      "name": "Mistral-7B-Instruct-v0.2",
      "provider": "Mistral AI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact version reported in Section 6 and Table 3.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Mistral-7B-Instruct-v0.2"
    },
    {
      "aliases": [
        "Moonshot"
      ],
      "entity_type": "model",
      "id": "proteinlmbench-moonshot",
      "name": "Moonshot (ProteinLMBench label)",
      "provider": "Not reported",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "The creator paper does not identify a provider checkpoint or API snapshot, and its citation is not sufficient to resolve the evaluated system.",
        "status": "verified"
      },
      "version_status": "not_reported",
      "version_string": null
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "proteinlmbench-qwen15-7b",
      "name": "Qwen1.5-7B",
      "provider": "Alibaba Cloud",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact creator-paper label.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Qwen1.5-7B"
    },
    {
      "aliases": [
        "Yi-6B"
      ],
      "entity_type": "model",
      "id": "proteinlmbench-yi-6b-chat",
      "name": "Yi-6B-Chat",
      "provider": "01.AI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Section 6 identifies Yi-6B-Chat; Table 3 shortens it to Yi-6B.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Yi-6B-Chat"
    },
    {
      "aliases": [
        "Protein MPNN"
      ],
      "entity_type": "model",
      "id": "proteinmpnn",
      "name": "ProteinMPNN",
      "provider": "University of Washington Institute for Protein Design",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact label in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "ProteinMPNN"
    },
    {
      "aliases": [
        "ProtGPT-2"
      ],
      "entity_type": "model",
      "id": "protgpt2",
      "name": "ProtGPT2",
      "provider": "University of Bayreuth",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact label in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "ProtGPT2"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "qwen-3-7-max",
      "name": "Qwen 3.7 Max",
      "provider": "Alibaba Cloud",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact GeneBench-Pro Supplementary Table 1 model label; endpoint snapshot and release date are not reported in the paper.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Qwen 3.7 Max"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "qwen-3-7-plus",
      "name": "Qwen 3.7 Plus",
      "provider": "Alibaba Cloud",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact GeneBench-Pro Supplementary Table 1 model label; endpoint snapshot and release date are not reported in the paper.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Qwen 3.7 Plus"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "rita-l",
      "name": "RITA L",
      "provider": "LightOn",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact size variant in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "RITA L"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "rita-m",
      "name": "RITA M",
      "provider": "LightOn",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact size variant in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "RITA M"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "rita-s",
      "name": "RITA S",
      "provider": "LightOn",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact size variant in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "RITA S"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "rita-xl",
      "name": "RITA XL",
      "provider": "LightOn",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact size variant in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "RITA XL"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "satija-et-al-seurat-disp",
      "name": "Seurat.disp",
      "provider": "Satija et al.",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-31",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "not_reported",
      "version_string": null
    },
    {
      "aliases": [
        "Gemini 2.5 Pro"
      ],
      "entity_type": "model",
      "id": "scbench-gemini-2-5-pro-unversioned",
      "name": "Gemini 2.5 Pro (scBench paper label)",
      "provider": "Google",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-27",
        "notes": "The scBench creator paper reports only the display label Gemini 2.5 Pro; it does not identify an exact API snapshot, so this record is not merged with the SCIGYM preview-specific model.",
        "status": "verified"
      },
      "version_status": "not_reported",
      "version_string": null
    },
    {
      "aliases": [
        "Claude-3.5-Haiku"
      ],
      "entity_type": "model",
      "id": "scigym-claude-3-5-haiku-20241022",
      "name": "Claude 3.5 Haiku 20241022 (SCIGYM)",
      "provider": "Anthropic",
      "release_date": "2024-10-22",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact dated API string in the creator paper and immutable SCIGYM evaluation artifact.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "claude-3-5-haiku-20241022"
    },
    {
      "aliases": [
        "Claude-3.7-Sonnet"
      ],
      "entity_type": "model",
      "id": "scigym-claude-3-7-sonnet-20250219",
      "name": "Claude 3.7 Sonnet 20250219 (SCIGYM)",
      "provider": "Anthropic",
      "release_date": "2025-02-19",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact dated API string in the creator paper and immutable SCIGYM evaluation artifact.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "claude-3-7-sonnet-20250219"
    },
    {
      "aliases": [
        "Gemini-2.5-Flash"
      ],
      "entity_type": "model",
      "id": "scigym-gemini-2-5-flash-preview-04-17",
      "name": "Gemini 2.5 Flash Preview 04-17 (SCIGYM)",
      "provider": "Google",
      "release_date": "2025-04-17",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact lowercase API string in the creator paper and immutable SCIGYM evaluation artifact.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "gemini-2.5-flash-preview-04-17"
    },
    {
      "aliases": [
        "Gemini-2.5-Pro"
      ],
      "entity_type": "model",
      "id": "scigym-gemini-2-5-pro-preview-03-25",
      "name": "Gemini 2.5 Pro Preview 03-25 (SCIGYM)",
      "provider": "Google",
      "release_date": "2025-03-25",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact lowercase API string in the creator paper and immutable SCIGYM evaluation artifact.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "gemini-2.5-pro-preview-03-25"
    },
    {
      "aliases": [
        "GPT-4.1"
      ],
      "entity_type": "model",
      "id": "scigym-gpt-4-1-2025-04-14",
      "name": "GPT-4.1 2025-04-14 (SCIGYM)",
      "provider": "OpenAI",
      "release_date": "2025-04-14",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact dated API string in the creator paper and immutable SCIGYM evaluation artifact.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "gpt-4.1-2025-04-14"
    },
    {
      "aliases": [
        "GPT-4.1-mini"
      ],
      "entity_type": "model",
      "id": "scigym-gpt-4-1-mini-2025-04-14",
      "name": "GPT-4.1 Mini 2025-04-14 (SCIGYM)",
      "provider": "OpenAI",
      "release_date": "2025-04-14",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact dated API string in the creator paper and immutable SCIGYM evaluation artifact.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "gpt-4.1-mini-2025-04-14"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "sirin-et-al-basa",
      "name": "bASA",
      "provider": "Sirin et al.",
      "release_date": null,
      "verification": {
        "last_verified": "2026-08-14",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "not_reported",
      "version_string": null
    },
    {
      "aliases": [
        "Site Independent"
      ],
      "entity_type": "model",
      "id": "site-independent",
      "name": "Site-Independent",
      "provider": "Harvard Medical School",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact baseline label in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Site-Independent ProteinGym baseline"
    },
    {
      "aliases": [
        "Grok-4.1"
      ],
      "entity_type": "model",
      "id": "spatialbench-grok-4-1-unversioned",
      "name": "Grok-4.1 (SpatialBench paper label)",
      "provider": "xAI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "The paper reports only Grok-4.1, so it is kept separate from the repository's exact grok-4-1-fast-reasoning identity.",
        "status": "verified"
      },
      "version_status": "not_reported",
      "version_string": null
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "stuart-et-al-seurat-vst",
      "name": "Seurat.vst",
      "provider": "Stuart et al.",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-31",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "not_reported",
      "version_string": null
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "tencent-hy-3-preview",
      "name": "Tencent HY 3 Preview",
      "provider": "Tencent",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact GeneBench-Pro Supplementary Table 1 model label; endpoint snapshot and release date are not reported in the paper.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Tencent HY 3 Preview"
    },
    {
      "aliases": [
        "TourSynbio-7B"
      ],
      "entity_type": "model",
      "id": "toursynbio-7b",
      "name": "InternLM2-Protein-7B",
      "provider": "TourSynbio",
      "release_date": "2024-06-08",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Creator-paper SSL-then-SFT system; a follow-up work renames the system TourSynbio-7B.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "InternLM2-Protein-7B"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "townes-et-al-deviancefs",
      "name": "devianceFS",
      "provider": "Townes et al.",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-31",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "not_reported",
      "version_string": null
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "trancepteve-l",
      "name": "TranceptEVE L",
      "provider": "University of Oxford",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact label in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "TranceptEVE L"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "trancepteve-m",
      "name": "TranceptEVE M",
      "provider": "University of Oxford",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact label in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "TranceptEVE M"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "trancepteve-s",
      "name": "TranceptEVE S",
      "provider": "University of Oxford",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact label in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "TranceptEVE S"
    },
    {
      "aliases": [
        "Tranception L"
      ],
      "entity_type": "model",
      "id": "tranception-l",
      "name": "Tranception L",
      "provider": "University of Oxford",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Table A5 label; retrieval is the default unless explicitly marked no retrieval.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Tranception L with retrieval"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "tranception-l-no-retrieval",
      "name": "Tranception L no retrieval",
      "provider": "University of Oxford",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact ablation label in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Tranception L no retrieval"
    },
    {
      "aliases": [
        "Tranception M"
      ],
      "entity_type": "model",
      "id": "tranception-m",
      "name": "Tranception M",
      "provider": "University of Oxford",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Table A5 label; retrieval is the default unless explicitly marked no retrieval.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Tranception M with retrieval"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "tranception-m-no-retrieval",
      "name": "Tranception M no retrieval",
      "provider": "University of Oxford",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact ablation label in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Tranception M no retrieval"
    },
    {
      "aliases": [
        "Tranception S"
      ],
      "entity_type": "model",
      "id": "tranception-s",
      "name": "Tranception S",
      "provider": "University of Oxford",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Table A5 label; retrieval is the default unless explicitly marked no retrieval.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Tranception S with retrieval"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "tranception-s-no-retrieval",
      "name": "Tranception S no retrieval",
      "provider": "University of Oxford",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact ablation label in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Tranception S no retrieval"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "unirep",
      "name": "UniRep",
      "provider": "Harvard University",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact label in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "UniRep"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "unirep-evotuned",
      "name": "UniRep evotuned",
      "provider": "Harvard University",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact result variant in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "UniRep evotuned"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "vespa",
      "name": "VESPA",
      "provider": "Technical University of Munich",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact label in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "VESPA"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "vespal",
      "name": "VESPAl",
      "provider": "Technical University of Munich",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact case-sensitive label in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "VESPAl"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "wavenet",
      "name": "WaveNet",
      "provider": "Harvard Medical School",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Exact label in ProteinGym v1.0 Table A5.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "WaveNet"
    },
    {
      "aliases": [],
      "entity_type": "model",
      "id": "xai-grok-4-20-reasoning",
      "name": "Grok 4.20 Reasoning",
      "provider": "xAI",
      "release_date": null,
      "verification": {
        "last_verified": "2026-07-30",
        "notes": "Exact model identity accepted by the automated double-pass paper review and pending owner PR approval.",
        "status": "verified"
      },
      "version_status": "reported",
      "version_string": "Grok 4.20 Reasoning"
    }
  ],
  "scientific_task_coverage": [
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "moleculenet",
      "benchmark_kind": "suite",
      "benchmark_name": "MoleculeNet",
      "benchmark_version": "original-2017",
      "classification_status": "partial",
      "confidence": "high",
      "count": 5,
      "count_basis": "Original paper dataset collections.",
      "count_ref": null,
      "count_unit": "other",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "enhancing-molecular-property-prediction-with-auxiliary",
        "moleculenet-paper"
      ],
      "evaluation_run_ids": [
        "moleculenet-creator-full"
      ],
      "evidence_ids": [
        "moleculenet-paper-definition-evidence"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "The five physiology collections are ESOL-independent ADMET or toxicity datasets; no endpoint-level total is asserted here.",
      "reporting_status": "reported",
      "root_family_id": "moleculenet",
      "task_type_id": "admet-toxicity-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "bioinstruction",
      "benchmark_kind": "suite",
      "benchmark_name": "Biology-Instructions",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 1,
      "count_basis": "Formal evaluation tracks (Antibody-Antigen Neutralization).",
      "count_ref": null,
      "count_unit": "tracks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "bioinstruction-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Track count; not an example count.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "antibody-antigen-interaction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "bioinstruction-aan",
      "benchmark_kind": "track",
      "benchmark_name": "Biology-Instructions Antibody-Antigen Neutralization",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 26902,
      "count_basis": "distinct examples across the published train, validation, and test splits",
      "count_ref": "/task_counts/total",
      "count_unit": "examples",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-aan-closed-baselines",
        "bioinstruction-aan-creator-systems",
        "bioinstruction-aan-open-baselines"
      ],
      "evidence_ids": [
        "bioinstruction-aan-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Antibody-antigen neutralization prediction.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "antibody-antigen-interaction"
    },
    {
      "aggregate_eligible": true,
      "as_of": "2026-07-21",
      "audit_status": "audited",
      "benchmark_id": "casp-immune-complexes",
      "benchmark_kind": "track",
      "benchmark_name": "CASP17 Immune Complexes",
      "benchmark_version": "CASP17",
      "classification_status": "complete",
      "confidence": "high",
      "count": null,
      "count_basis": "CASP17 immune-complex targets.",
      "count_ref": null,
      "count_unit": "targets",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "casp-immune-evidence-category"
      ],
      "mapping_method": "official-track",
      "notes": "The official current-round page does not publish a standalone count.",
      "reporting_status": "not_reported",
      "root_family_id": "casp",
      "task_type_id": "antibody-antigen-interaction"
    },
    {
      "aggregate_eligible": true,
      "as_of": "2026-06-10",
      "audit_status": "audited-with-caveats",
      "benchmark_id": "spatialbench",
      "benchmark_kind": "agentic-eval",
      "benchmark_name": "SpatialBench",
      "benchmark_version": "repo-159-5042c4f",
      "classification_status": "partial",
      "confidence": "high",
      "count": 3,
      "count_basis": "official category_results.json n_evals",
      "count_ref": "/task_counts/subsets/1/count",
      "count_unit": "problems",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "spatialbench-preprint",
        "spatialbench-repository-release",
        "system-card-claude-opus-5"
      ],
      "evaluation_run_ids": [
        "spatialbench-paper-v2-base",
        "spatialbench-paper-v2-claude-code",
        "spatialbench-paper-v2-latch",
        "spatialbench-repo-159-claude-code",
        "spatialbench-repo-159-mini-swe-agent",
        "spatialbench-repo-159-openai-codex",
        "spatialbench-repo-159-pi"
      ],
      "evidence_ids": [
        "spatialbench-evidence-current-counts"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "Official Clustering category.",
      "reporting_status": "reported",
      "root_family_id": "spatialbench",
      "task_type_id": "cell-state-clustering"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "single-cell-omics-arena-soar",
      "benchmark_kind": "suite",
      "benchmark_name": "Single-cell Omics Arena",
      "benchmark_version": "initial-release",
      "classification_status": "partial",
      "confidence": "high",
      "count": 1226,
      "count_basis": "Benchmark-wide cell-type annotation task total reported by the creator paper.",
      "count_ref": null,
      "count_unit": "tasks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "single-cell-omics-arena-soar-repository-result-snapshot"
      ],
      "evaluation_run_ids": [
        "soar-e5d2b3e-rna-zero-shot",
        "soar-e5d2b3e-rna-zero-shot-cot"
      ],
      "evidence_ids": [
        "single-cell-omics-arena-soar-automated-task-1-evidence"
      ],
      "mapping_method": "official-taxonomy",
      "notes": null,
      "reporting_status": "reported",
      "root_family_id": "single-cell-omics-arena-soar",
      "task_type_id": "cell-type-annotation"
    },
    {
      "aggregate_eligible": true,
      "as_of": "2026-06-10",
      "audit_status": "audited-with-caveats",
      "benchmark_id": "spatialbench",
      "benchmark_kind": "agentic-eval",
      "benchmark_name": "SpatialBench",
      "benchmark_version": "repo-159-5042c4f",
      "classification_status": "partial",
      "confidence": "high",
      "count": 45,
      "count_basis": "official category_results.json n_evals",
      "count_ref": "/task_counts/subsets/0/count",
      "count_unit": "problems",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "spatialbench-preprint",
        "spatialbench-repository-release",
        "system-card-claude-opus-5"
      ],
      "evaluation_run_ids": [
        "spatialbench-paper-v2-base",
        "spatialbench-paper-v2-claude-code",
        "spatialbench-paper-v2-latch",
        "spatialbench-repo-159-claude-code",
        "spatialbench-repo-159-mini-swe-agent",
        "spatialbench-repo-159-openai-codex",
        "spatialbench-repo-159-pi"
      ],
      "evidence_ids": [
        "spatialbench-evidence-current-counts"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "Official Cell Typing category.",
      "reporting_status": "reported",
      "root_family_id": "spatialbench",
      "task_type_id": "cell-type-annotation"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "beacon-rna",
      "benchmark_kind": "suite",
      "benchmark_name": "BEACON",
      "benchmark_version": "neurips-2024",
      "classification_status": "complete",
      "confidence": "high",
      "count": 1,
      "count_basis": "Formal RNA benchmark tasks.",
      "count_ref": null,
      "count_unit": "tasks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "beacon-paper"
      ],
      "evaluation_run_ids": [
        "beacon-creator-full"
      ],
      "evidence_ids": [
        "beacon-paper-definition-evidence"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "CRISPR on-target efficiency prediction.",
      "reporting_status": "reported",
      "root_family_id": "beacon-rna",
      "task_type_id": "crispr-guide-activity-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "bioinstruction",
      "benchmark_kind": "suite",
      "benchmark_name": "Biology-Instructions",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 1,
      "count_basis": "Formal evaluation tracks (CRISPR On-Target Prediction).",
      "count_ref": null,
      "count_unit": "tracks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "bioinstruction-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Track count; not an example count.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "crispr-guide-activity-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "bioinstruction-crispr-on-target",
      "benchmark_kind": "track",
      "benchmark_name": "Biology-Instructions CRISPR On-Target Prediction",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 2076,
      "count_basis": "distinct examples across the published train, validation, and test splits",
      "count_ref": "/task_counts/total",
      "count_unit": "examples",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-crispr-on-target-closed-baselines",
        "bioinstruction-crispr-on-target-creator-systems",
        "bioinstruction-crispr-on-target-open-baselines"
      ],
      "evidence_ids": [
        "bioinstruction-crispr-on-target-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "CRISPR guide on-target activity prediction.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "crispr-guide-activity-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "beacon-rna",
      "benchmark_kind": "suite",
      "benchmark_name": "BEACON",
      "benchmark_version": "neurips-2024",
      "classification_status": "complete",
      "confidence": "high",
      "count": 1,
      "count_basis": "Formal RNA benchmark tasks.",
      "count_ref": null,
      "count_unit": "tasks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "beacon-paper"
      ],
      "evaluation_run_ids": [
        "beacon-creator-full"
      ],
      "evidence_ids": [
        "beacon-paper-definition-evidence"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "CRISPR off-target interaction prediction.",
      "reporting_status": "reported",
      "root_family_id": "beacon-rna",
      "task_type_id": "crispr-off-target-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": "2026-06-10",
      "audit_status": "audited-with-caveats",
      "benchmark_id": "spatialbench",
      "benchmark_kind": "agentic-eval",
      "benchmark_name": "SpatialBench",
      "benchmark_version": "repo-159-5042c4f",
      "classification_status": "partial",
      "confidence": "high",
      "count": 40,
      "count_basis": "official category_results.json n_evals",
      "count_ref": "/task_counts/subsets/2/count",
      "count_unit": "problems",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "spatialbench-preprint",
        "spatialbench-repository-release",
        "system-card-claude-opus-5"
      ],
      "evaluation_run_ids": [
        "spatialbench-paper-v2-base",
        "spatialbench-paper-v2-claude-code",
        "spatialbench-paper-v2-latch",
        "spatialbench-repo-159-claude-code",
        "spatialbench-repo-159-mini-swe-agent",
        "spatialbench-repo-159-openai-codex",
        "spatialbench-repo-159-pi"
      ],
      "evidence_ids": [
        "spatialbench-evidence-current-counts"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "Official Differential Expression category.",
      "reporting_status": "reported",
      "root_family_id": "spatialbench",
      "task_type_id": "differential-expression-analysis"
    },
    {
      "aggregate_eligible": true,
      "as_of": "2026-07-22",
      "audit_status": "audited",
      "benchmark_id": "genomic-benchmarks",
      "benchmark_kind": "suite",
      "benchmark_name": "Genomic Benchmarks",
      "benchmark_version": "package-1.0.0-snapshot",
      "classification_status": "complete",
      "confidence": "high",
      "count": 3,
      "count_basis": "Benchmark dataset classification tasks.",
      "count_ref": null,
      "count_unit": "tasks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "genomic-benchmarks-paper"
      ],
      "evaluation_run_ids": [
        "genomic-benchmarks-creator-full"
      ],
      "evidence_ids": [
        "genomic-benchmarks-paper-definition-evidence"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "Two demo coding-versus-intergenic datasets and the human regulatory-region multiclass dataset.",
      "reporting_status": "reported",
      "root_family_id": "genomic-benchmarks",
      "task_type_id": "dna-sequence-analysis"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-seqqa-pcr-geneprimers-enz",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench SeqQA — Primers-to-restriction enzymes",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "complete",
      "confidence": "high",
      "count": 50,
      "count_basis": "questions across public and private splits",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-seqqa-pcr-geneprimers-enz-creator-mcq"
      ],
      "evidence_ids": [
        "lab-bench-seqqa-pcr-geneprimers-enz-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Infers restriction enzymes from a primer pair rather than generating a new sequence.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "dna-sequence-analysis"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-seqqa-pcr-primers-len",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench SeqQA — Primers to amplicon length",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "complete",
      "confidence": "high",
      "count": 50,
      "count_basis": "questions across public and private splits",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-seqqa-pcr-primers-len-creator-mcq"
      ],
      "evidence_ids": [
        "lab-bench-seqqa-pcr-primers-len-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Calculates an amplicon length from primers and a DNA template.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "dna-sequence-analysis"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-seqqa-prop-seq-gcpercent",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench SeqQA — GC percentage",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "complete",
      "confidence": "high",
      "count": 50,
      "count_basis": "questions across public and private splits",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-seqqa-prop-seq-gcpercent-creator-mcq"
      ],
      "evidence_ids": [
        "lab-bench-seqqa-prop-seq-gcpercent-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Calculates a DNA sequence GC percentage.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "dna-sequence-analysis"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-seqqa-re-seq-lenfrags",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench SeqQA — Restriction-fragment lengths",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "complete",
      "confidence": "high",
      "count": 50,
      "count_basis": "questions across public and private splits",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-seqqa-re-seq-lenfrags-creator-mcq"
      ],
      "evidence_ids": [
        "lab-bench-seqqa-re-seq-lenfrags-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Calculates restriction-fragment lengths.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "dna-sequence-analysis"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-seqqa-re-seq-numfrags",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench SeqQA — Restriction-fragment count",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "complete",
      "confidence": "high",
      "count": 50,
      "count_basis": "questions across public and private splits",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-seqqa-re-seq-numfrags-creator-mcq"
      ],
      "evidence_ids": [
        "lab-bench-seqqa-re-seq-numfrags-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Calculates a restriction-fragment count.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "dna-sequence-analysis"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-seqqa",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench SeqQA",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "partial",
      "confidence": "high",
      "count": null,
      "count_basis": "Primer-design questions across formal SeqQA child tracks.",
      "count_ref": null,
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "anthropic-sonnet-4-5-system-card"
      ],
      "evaluation_run_ids": [
        "lab-bench-seqqa-anthropic-sonnet45-system-card"
      ],
      "evidence_ids": [
        "lab-bench-seqqa-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "A cross-track subtotal is not reported by the creator.",
      "reporting_status": "not_reported",
      "root_family_id": "lab-bench",
      "task_type_id": "dna-sequence-design"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-seqqa-pcr-gene-enzprimers",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench SeqQA — Gene-to-restriction primers",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "complete",
      "confidence": "high",
      "count": 50,
      "count_basis": "questions across public and private splits",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-seqqa-pcr-gene-enzprimers-creator-mcq"
      ],
      "evidence_ids": [
        "lab-bench-seqqa-pcr-gene-enzprimers-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Restriction-cloning primer design.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "dna-sequence-design"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-seqqa-pcr-gene-gibshindprimers",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench SeqQA — Gene-to-Gibson primers (HindIII)",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "complete",
      "confidence": "high",
      "count": 50,
      "count_basis": "questions across public and private splits",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-seqqa-pcr-gene-gibshindprimers-creator-mcq"
      ],
      "evidence_ids": [
        "lab-bench-seqqa-pcr-gene-gibshindprimers-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Gibson-assembly primer design.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "dna-sequence-design"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-seqqa-pcr-gene-gibssmaprimers",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench SeqQA — Gene-to-Gibson primers (SmaI)",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "complete",
      "confidence": "high",
      "count": 50,
      "count_basis": "questions across public and private splits",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-seqqa-pcr-gene-gibssmaprimers-creator-mcq"
      ],
      "evidence_ids": [
        "lab-bench-seqqa-pcr-gene-gibssmaprimers-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Gibson-assembly primer design.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "dna-sequence-design"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-seqqa-pcr-len-primers",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench SeqQA — Amplicon length to primers",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "complete",
      "confidence": "high",
      "count": 50,
      "count_basis": "questions across public and private splits",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-seqqa-pcr-len-primers-creator-mcq"
      ],
      "evidence_ids": [
        "lab-bench-seqqa-pcr-len-primers-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "PCR primer selection under an amplicon-length constraint.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "dna-sequence-design"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-seqqa-pcr-seq-enzprimers",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench SeqQA — Sequence-to-restriction primers",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "complete",
      "confidence": "high",
      "count": 50,
      "count_basis": "questions across public and private splits",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-seqqa-pcr-seq-enzprimers-creator-mcq"
      ],
      "evidence_ids": [
        "lab-bench-seqqa-pcr-seq-enzprimers-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Restriction-cloning primer design from a sequence.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "dna-sequence-design"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-seqqa-pcr-seq-primers",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench SeqQA — Amplicon sequence to primers",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "complete",
      "confidence": "high",
      "count": 50,
      "count_basis": "questions across public and private splits",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-seqqa-pcr-seq-primers-creator-mcq"
      ],
      "evidence_ids": [
        "lab-bench-seqqa-pcr-seq-primers-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "PCR primer selection for a target amplicon.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "dna-sequence-design"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "biomysterybench",
      "benchmark_kind": "agentic-eval",
      "benchmark_name": "BioMysteryBench",
      "benchmark_version": "v11",
      "classification_status": "partial",
      "confidence": "high",
      "count": 90,
      "count_basis": "v11 mystery-bioinformatics problems after the June 2026 answer-key audit",
      "count_ref": "/task_counts/total",
      "count_unit": "problems",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "biomysterybench-official",
        "system-card-claude-opus-5"
      ],
      "evaluation_run_ids": [
        "biomysterybench-official-run",
        "biomysterybench-v8-human-difficult",
        "biomysterybench-v8-human-solvable"
      ],
      "evidence_ids": [
        "biomysterybench-evidence-v11"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "Each mystery is scored on its final answer rather than a prescribed analysis path.",
      "reporting_status": "reported",
      "root_family_id": "biomysterybench",
      "task_type_id": "end-to-end-computational-analysis"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "bixbench",
      "benchmark_kind": "agentic-eval",
      "benchmark_name": "BixBench",
      "benchmark_version": "v1.5",
      "classification_status": "partial",
      "confidence": "high",
      "count": 205,
      "count_basis": "one question per row in the official v1.5 BixBench.jsonl",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "anthropic-life-sciences",
        "bixbench-paper",
        "bixbench-v1-5-release"
      ],
      "evaluation_run_ids": [
        "bixbench-creator-paper",
        "bixbench-paper-mcq-no-images",
        "bixbench-paper-mcq-no-refusal",
        "bixbench-paper-mcq-refusal",
        "bixbench-v1-5-agentic-mcq-no-refusal-images",
        "bixbench-v1-5-agentic-mcq-refusal-images",
        "bixbench-v1-5-agentic-mcq-refusal-no-images",
        "bixbench-v1-5-agentic-open-images",
        "bixbench-v1-5-zero-shot-mcq-no-refusal",
        "bixbench-v1-5-zero-shot-mcq-refusal",
        "bixbench-v1-5-zero-shot-open"
      ],
      "evidence_ids": [
        "bixbench-evidence-v1-5-counts"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "Questions are grounded in containerized published analysis capsules.",
      "reporting_status": "reported",
      "root_family_id": "bixbench",
      "task_type_id": "end-to-end-computational-analysis"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "blade",
      "benchmark_kind": "suite",
      "benchmark_name": "BLADE",
      "benchmark_version": "arXiv v3",
      "classification_status": "partial",
      "confidence": "high",
      "count": 12,
      "count_basis": "paired real-world research questions and datasets used as the source units for BLADE",
      "count_ref": "/task_counts/total",
      "count_unit": "problems",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "blade-evidence-current-counts"
      ],
      "mapping_method": "official-track",
      "notes": "Coverage is suite-wide; only four source questions are in life science.",
      "reporting_status": "reported",
      "root_family_id": "blade",
      "task_type_id": "end-to-end-computational-analysis"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "blade-analysis-generation",
      "benchmark_kind": "track",
      "benchmark_name": "BLADE End-to-End Analysis Generation",
      "benchmark_version": "arXiv v3",
      "classification_status": "complete",
      "confidence": "high",
      "count": 12,
      "count_basis": "paired research questions and datasets requiring a complete generated analysis",
      "count_ref": "/task_counts/total",
      "count_unit": "problems",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "blade-paper"
      ],
      "evaluation_run_ids": [
        "blade-creator-paper",
        "blade-creator-react"
      ],
      "evidence_ids": [
        "blade-generation-evidence-counts"
      ],
      "mapping_method": "official-track",
      "notes": "The track grades complete analysis decisions and execution.",
      "reporting_status": "reported",
      "root_family_id": "blade",
      "task_type_id": "end-to-end-computational-analysis"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "compbiobench",
      "benchmark_kind": "agentic-eval",
      "benchmark_name": "CompBioBench",
      "benchmark_version": "v1",
      "classification_status": "partial",
      "confidence": "high",
      "count": 100,
      "count_basis": "v1 independent computational-biology tasks",
      "count_ref": "/task_counts/total",
      "count_unit": "tasks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "compbiobench-preprint"
      ],
      "evaluation_run_ids": [
        "compbiobench-codex-hardest",
        "compbiobench-creator-full",
        "compbiobench-haiku-full",
        "compbiobench-haiku-hardest",
        "compbiobench-nonagentic-baselines",
        "compbiobench-opus-full",
        "compbiobench-opus-hardest",
        "compbiobench-sonnet-full",
        "compbiobench-sonnet-hardest"
      ],
      "evidence_ids": [
        "compbiobench-evidence-counts",
        "compbiobench-evidence-runner-license"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "Tasks require code, tools, and external resources.",
      "reporting_status": "reported",
      "root_family_id": "compbiobench",
      "task_type_id": "end-to-end-computational-analysis"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "genebench-pro",
      "benchmark_kind": "agentic-eval",
      "benchmark_name": "GeneBench-Pro",
      "benchmark_version": "paper-v1",
      "classification_status": "partial",
      "confidence": "high",
      "count": 129,
      "count_basis": "self-contained synthetic scientific-analysis problems (called evaluations in the paper abstract)",
      "count_ref": "/task_counts/total",
      "count_unit": "problems",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "genebench-pro-report"
      ],
      "evaluation_run_ids": [
        "genebench-pro-claude-high",
        "genebench-pro-claude-low",
        "genebench-pro-claude-max",
        "genebench-pro-claude-medium",
        "genebench-pro-claude-xhigh",
        "genebench-pro-official",
        "genebench-pro-pro-mode",
        "genebench-pro-reasoning-enabled",
        "genebench-pro-standard-high",
        "genebench-pro-standard-low",
        "genebench-pro-standard-max",
        "genebench-pro-standard-medium",
        "genebench-pro-standard-none"
      ],
      "evidence_ids": [
        "genebench-pro-paper-evidence"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "Problems require multistage analysis, diagnostics, and judgment.",
      "reporting_status": "reported",
      "root_family_id": "genebench-pro",
      "task_type_id": "end-to-end-computational-analysis"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "bioinstruction",
      "benchmark_kind": "suite",
      "benchmark_name": "Biology-Instructions",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 1,
      "count_basis": "Formal evaluation tracks (Enhancer Activity Prediction).",
      "count_ref": null,
      "count_unit": "tracks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "bioinstruction-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Track count; not an example count.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "enhancer-activity-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "bioinstruction-ea",
      "benchmark_kind": "track",
      "benchmark_name": "Biology-Instructions Enhancer Activity Prediction",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 484052,
      "count_basis": "distinct examples across the published train, validation, and test splits",
      "count_ref": "/task_counts/total",
      "count_unit": "examples",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-ea-closed-baselines",
        "bioinstruction-ea-creator-systems",
        "bioinstruction-ea-open-baselines"
      ],
      "evidence_ids": [
        "bioinstruction-ea-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Enhancer activity prediction.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "enhancer-activity-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": "2026-07-22",
      "audit_status": "audited",
      "benchmark_id": "genomic-benchmarks",
      "benchmark_kind": "suite",
      "benchmark_name": "Genomic Benchmarks",
      "benchmark_version": "package-1.0.0-snapshot",
      "classification_status": "complete",
      "confidence": "high",
      "count": 4,
      "count_basis": "Benchmark dataset classification tasks.",
      "count_ref": null,
      "count_unit": "tasks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "genomic-benchmarks-paper"
      ],
      "evaluation_run_ids": [
        "genomic-benchmarks-creator-full"
      ],
      "evidence_ids": [
        "genomic-benchmarks-paper-definition-evidence"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "Mouse enhancer, Drosophila enhancer, human enhancer Cohn, and human enhancer Ensembl datasets.",
      "reporting_status": "reported",
      "root_family_id": "genomic-benchmarks",
      "task_type_id": "enhancer-activity-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "bioinstruction",
      "benchmark_kind": "suite",
      "benchmark_name": "Biology-Instructions",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 1,
      "count_basis": "Formal evaluation tracks (Enhancer-Promoter Interaction Prediction).",
      "count_ref": null,
      "count_unit": "tracks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "bioinstruction-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Track count; not an example count.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "enhancer-promoter-interaction-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "bioinstruction-epi",
      "benchmark_kind": "track",
      "benchmark_name": "Biology-Instructions Enhancer-Promoter Interaction Prediction",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 16368,
      "count_basis": "distinct examples across the published train, validation, and test splits",
      "count_ref": "/task_counts/total",
      "count_unit": "examples",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-epi-closed-baselines",
        "bioinstruction-epi-creator-systems",
        "bioinstruction-epi-open-baselines"
      ],
      "evidence_ids": [
        "bioinstruction-epi-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Enhancer-promoter interaction prediction.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "enhancer-promoter-interaction-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "bioinstruction",
      "benchmark_kind": "suite",
      "benchmark_name": "Biology-Instructions",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 1,
      "count_basis": "Formal evaluation tracks (Epigenetic Marks Prediction).",
      "count_ref": null,
      "count_unit": "tracks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "bioinstruction-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Track count; not an example count.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "epigenetic-mark-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "bioinstruction-emp",
      "benchmark_kind": "track",
      "benchmark_name": "Biology-Instructions Epigenetic Marks Prediction",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 287367,
      "count_basis": "distinct examples across the published train, validation, and test splits",
      "count_ref": "/task_counts/total",
      "count_unit": "examples",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-emp-closed-baselines",
        "bioinstruction-emp-creator-systems",
        "bioinstruction-emp-open-baselines"
      ],
      "evidence_ids": [
        "bioinstruction-emp-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Epigenetic-mark prediction.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "epigenetic-mark-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": "2026-07-22",
      "audit_status": "audited",
      "benchmark_id": "genomic-benchmarks",
      "benchmark_kind": "suite",
      "benchmark_name": "Genomic Benchmarks",
      "benchmark_version": "package-1.0.0-snapshot",
      "classification_status": "complete",
      "confidence": "high",
      "count": 1,
      "count_basis": "Benchmark dataset classification tasks.",
      "count_ref": null,
      "count_unit": "tasks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "genomic-benchmarks-paper"
      ],
      "evaluation_run_ids": [
        "genomic-benchmarks-creator-full"
      ],
      "evidence_ids": [
        "genomic-benchmarks-paper-definition-evidence"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "Human open-chromatin region classification.",
      "reporting_status": "reported",
      "root_family_id": "genomic-benchmarks",
      "task_type_id": "epigenetic-mark-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "lab-bench",
      "benchmark_kind": "suite",
      "benchmark_name": "LAB-Bench",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "partial",
      "confidence": "high",
      "count": 135,
      "count_basis": "ProtocolQA questions across public and private splits.",
      "count_ref": null,
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "lab-bench-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "CloningScenarios is shown separately and is not included in this count.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "experiment-protocol-planning"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-cloning-scenarios",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench CloningScenarios",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "complete",
      "confidence": "high",
      "count": 41,
      "count_basis": "questions across public and private splits",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "anthropic-sonnet-4-5-system-card",
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-cloning-scenarios-anthropic-sonnet45-system-card",
        "lab-bench-cloning-scenarios-creator-mcq",
        "lab-bench-cloning-scenarios-creator-mcq-llama-context",
        "lab-bench-cloning-scenarios-creator-open-response"
      ],
      "evidence_ids": [
        "lab-bench-cloning-scenarios-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Molecular-cloning workflow planning.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "experiment-protocol-planning"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-protocolqa",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench ProtocolQA",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "complete",
      "confidence": "high",
      "count": 135,
      "count_basis": "questions across public and private splits",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "anthropic-life-sciences",
        "anthropic-sonnet-4-5-system-card",
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-protocolqa-anthropic",
        "lab-bench-protocolqa-anthropic-sonnet45-system-card",
        "lab-bench-protocolqa-creator-mcq",
        "lab-bench-protocolqa-creator-open-response"
      ],
      "evidence_ids": [
        "lab-bench-protocolqa-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Protocol troubleshooting is classified under protocol planning.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "experiment-protocol-planning"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "proteinlmbench",
      "benchmark_kind": "dataset",
      "benchmark_name": "ProteinLMBench",
      "benchmark_version": "hf-f139796",
      "classification_status": "partial",
      "confidence": "high",
      "count": null,
      "count_basis": "Released ProteinLMBench question records.",
      "count_ref": null,
      "count_unit": "questions",
      "coverage": "observed",
      "evaluating_work_ids": [
        "proteinlmbench-paper"
      ],
      "evaluation_run_ids": [
        "proteinlmbench-creator-full"
      ],
      "evidence_ids": [
        "proteinlmbench-evidence-paper"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "Official questions discuss multiple binding contexts without a topic field or binding-subtype counts.",
      "reporting_status": "not_reported",
      "root_family_id": "proteinlmbench",
      "task_type_id": "molecular-interaction-analysis"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "biomysterybench",
      "benchmark_kind": "agentic-eval",
      "benchmark_name": "BioMysteryBench",
      "benchmark_version": "v11",
      "classification_status": "partial",
      "confidence": "high",
      "count": null,
      "count_basis": "v11 mystery-bioinformatics problems using omics and cellular data.",
      "count_ref": null,
      "count_unit": "problems",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "biomysterybench-official",
        "system-card-claude-opus-5"
      ],
      "evaluation_run_ids": [
        "biomysterybench-official-run",
        "biomysterybench-v8-human-difficult",
        "biomysterybench-v8-human-solvable"
      ],
      "evidence_ids": [
        "biomysterybench-evidence-taxonomy"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "Single-cell, proteomics, metabolomics, and other modalities are explicit, without a leaf-task count.",
      "reporting_status": "not_reported",
      "root_family_id": "biomysterybench",
      "task_type_id": "omics-cellular-analysis"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "bixbench",
      "benchmark_kind": "agentic-eval",
      "benchmark_name": "BixBench",
      "benchmark_version": "v1.5",
      "classification_status": "partial",
      "confidence": "high",
      "count": null,
      "count_basis": "v1.5 questions carrying official genomics, transcriptomics, epigenomics, single-cell, proteomics, or integrative-omics category labels.",
      "count_ref": null,
      "count_unit": "questions",
      "coverage": "observed",
      "evaluating_work_ids": [
        "anthropic-life-sciences",
        "bixbench-paper",
        "bixbench-v1-5-release"
      ],
      "evaluation_run_ids": [
        "bixbench-creator-paper",
        "bixbench-paper-mcq-no-images",
        "bixbench-paper-mcq-no-refusal",
        "bixbench-paper-mcq-refusal",
        "bixbench-v1-5-agentic-mcq-no-refusal-images",
        "bixbench-v1-5-agentic-mcq-refusal-images",
        "bixbench-v1-5-agentic-mcq-refusal-no-images",
        "bixbench-v1-5-agentic-open-images",
        "bixbench-v1-5-zero-shot-mcq-no-refusal",
        "bixbench-v1-5-zero-shot-mcq-refusal",
        "bixbench-v1-5-zero-shot-open"
      ],
      "evidence_ids": [
        "bixbench-evidence-taxonomy"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "Official multi-label domain categories overlap, so no additive omics question total is asserted.",
      "reporting_status": "not_reported",
      "root_family_id": "bixbench",
      "task_type_id": "omics-cellular-analysis"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "compbiobench",
      "benchmark_kind": "agentic-eval",
      "benchmark_name": "CompBioBench",
      "benchmark_version": "v1",
      "classification_status": "partial",
      "confidence": "high",
      "count": null,
      "count_basis": "v1 tasks in official genomics, transcriptomics, epigenetics, single-cell, and spatial domains.",
      "count_ref": null,
      "count_unit": "tasks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "compbiobench-preprint"
      ],
      "evaluation_run_ids": [
        "compbiobench-codex-hardest",
        "compbiobench-creator-full",
        "compbiobench-haiku-full",
        "compbiobench-haiku-hardest",
        "compbiobench-nonagentic-baselines",
        "compbiobench-opus-full",
        "compbiobench-opus-hardest",
        "compbiobench-sonnet-full",
        "compbiobench-sonnet-hardest"
      ],
      "evidence_ids": [
        "compbiobench-evidence-taxonomy"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "The official domains establish broad omics coverage but do not provide an exhaustive Scientific Task breakdown.",
      "reporting_status": "not_reported",
      "root_family_id": "compbiobench",
      "task_type_id": "omics-cellular-analysis"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "genebench-pro",
      "benchmark_kind": "agentic-eval",
      "benchmark_name": "GeneBench-Pro",
      "benchmark_version": "paper-v1",
      "classification_status": "partial",
      "confidence": "high",
      "count": null,
      "count_basis": "Problems assigned to transcriptomics, epigenomics, single-cell, spatial, proteomics, microbiome, and related official domains.",
      "count_ref": null,
      "count_unit": "problems",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "genebench-pro-report"
      ],
      "evaluation_run_ids": [
        "genebench-pro-claude-high",
        "genebench-pro-claude-low",
        "genebench-pro-claude-max",
        "genebench-pro-claude-medium",
        "genebench-pro-claude-xhigh",
        "genebench-pro-official",
        "genebench-pro-pro-mode",
        "genebench-pro-reasoning-enabled",
        "genebench-pro-standard-high",
        "genebench-pro-standard-low",
        "genebench-pro-standard-max",
        "genebench-pro-standard-medium",
        "genebench-pro-standard-none"
      ],
      "evidence_ids": [
        "genebench-pro-paper-evidence"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "The official domain atlas supports broad omics coverage but not an exhaustive leaf-task subtotal.",
      "reporting_status": "not_reported",
      "root_family_id": "genebench-pro",
      "task_type_id": "omics-cellular-analysis"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "bioinstruction",
      "benchmark_kind": "suite",
      "benchmark_name": "Biology-Instructions",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 2,
      "count_basis": "Formal evaluation tracks (Core Promoter Detection and Promoter Detection 300).",
      "count_ref": null,
      "count_unit": "tracks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "bioinstruction-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Track count; not an example count.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "promoter-detection"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "bioinstruction-cpd",
      "benchmark_kind": "track",
      "benchmark_name": "Biology-Instructions Core Promoter Detection",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 118392,
      "count_basis": "distinct examples across the published train, validation, and test splits",
      "count_ref": "/task_counts/total",
      "count_unit": "examples",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-cpd-closed-baselines",
        "bioinstruction-cpd-creator-systems",
        "bioinstruction-cpd-open-baselines"
      ],
      "evidence_ids": [
        "bioinstruction-cpd-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Core-promoter detection.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "promoter-detection"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "bioinstruction-pd300",
      "benchmark_kind": "track",
      "benchmark_name": "Biology-Instructions Promoter Detection 300",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 118392,
      "count_basis": "distinct examples across the published train, validation, and test splits",
      "count_ref": "/task_counts/total",
      "count_unit": "examples",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-pd300-closed-baselines",
        "bioinstruction-pd300-creator-systems",
        "bioinstruction-pd300-open-baselines"
      ],
      "evidence_ids": [
        "bioinstruction-pd300-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Promoter detection in 300-base-pair context.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "promoter-detection"
    },
    {
      "aggregate_eligible": true,
      "as_of": "2026-07-22",
      "audit_status": "audited",
      "benchmark_id": "genomic-benchmarks",
      "benchmark_kind": "suite",
      "benchmark_name": "Genomic Benchmarks",
      "benchmark_version": "package-1.0.0-snapshot",
      "classification_status": "complete",
      "confidence": "high",
      "count": 1,
      "count_basis": "Benchmark dataset classification tasks.",
      "count_ref": null,
      "count_unit": "tasks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "genomic-benchmarks-paper"
      ],
      "evaluation_run_ids": [
        "genomic-benchmarks-creator-full"
      ],
      "evidence_ids": [
        "genomic-benchmarks-paper-definition-evidence"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "Human non-TATA promoter classification.",
      "reporting_status": "reported",
      "root_family_id": "genomic-benchmarks",
      "task_type_id": "promoter-detection"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-dbqa-variant-from-sequence",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench DbQA — Protein variant from sequence",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "complete",
      "confidence": "high",
      "count": 100,
      "count_basis": "questions across public and private splits",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-dbqa-variant-from-sequence-creator-mcq"
      ],
      "evidence_ids": [
        "lab-bench-dbqa-variant-from-sequence-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Variant pathogenicity is explicitly evaluated.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "protein-clinical-variant-interpretation"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-dbqa-variant-multi-sequence",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench DbQA — Protein variant with multiple sequences",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "complete",
      "confidence": "high",
      "count": 100,
      "count_basis": "questions across public and private splits",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-dbqa-variant-multi-sequence-creator-mcq"
      ],
      "evidence_ids": [
        "lab-bench-dbqa-variant-multi-sequence-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Variant pathogenicity is explicitly evaluated.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "protein-clinical-variant-interpretation"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "proteingym",
      "benchmark_kind": "suite",
      "benchmark_name": "ProteinGym",
      "benchmark_version": "1.3",
      "classification_status": "partial",
      "confidence": "high",
      "count": null,
      "count_basis": "Clinical substitution and indel tracks.",
      "count_ref": null,
      "count_unit": "records",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "system-card-claude-opus-5"
      ],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "proteingym-evidence-taxonomy"
      ],
      "mapping_method": "official-track",
      "notes": "Child records use clinical proteins as their count unit.",
      "reporting_status": "not_reported",
      "root_family_id": "proteingym",
      "task_type_id": "protein-clinical-variant-interpretation"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "proteingym-clinical-indels",
      "benchmark_kind": "track",
      "benchmark_name": "ProteinGym Clinical Indels",
      "benchmark_version": "1.3",
      "classification_status": "complete",
      "confidence": "high",
      "count": 1555,
      "count_basis": "Clinical proteins.",
      "count_ref": "/task_counts/total",
      "count_unit": "records",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "proteingym-clinical-indel-evidence-v13"
      ],
      "mapping_method": "official-track",
      "notes": "The registry count unit is proteins, not variants.",
      "reporting_status": "reported",
      "root_family_id": "proteingym",
      "task_type_id": "protein-clinical-variant-interpretation"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "proteingym-clinical-substitutions",
      "benchmark_kind": "track",
      "benchmark_name": "ProteinGym Clinical Substitutions",
      "benchmark_version": "1.3",
      "classification_status": "complete",
      "confidence": "high",
      "count": 2525,
      "count_basis": "Clinical proteins.",
      "count_ref": "/task_counts/total",
      "count_unit": "records",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "proteingym-clinical-sub-evidence-v13"
      ],
      "mapping_method": "official-track",
      "notes": "The registry count unit is proteins, not variants.",
      "reporting_status": "reported",
      "root_family_id": "proteingym",
      "task_type_id": "protein-clinical-variant-interpretation"
    },
    {
      "aggregate_eligible": true,
      "as_of": "2026-07-22",
      "audit_status": "audited",
      "benchmark_id": "atom3d",
      "benchmark_kind": "suite",
      "benchmark_name": "ATOM3D",
      "benchmark_version": "v0.2.6",
      "classification_status": "complete",
      "confidence": "high",
      "count": 1,
      "count_basis": "Curated 3D benchmark datasets.",
      "count_ref": null,
      "count_unit": "tasks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "atom3d-paper"
      ],
      "evaluation_run_ids": [
        "atom3d-creator-full"
      ],
      "evidence_ids": [
        "atom3d-paper-definition-evidence"
      ],
      "mapping_method": "official-track",
      "notes": "MSP predicts whether a single mutation stabilizes a protein complex; it is not a monomer thermostability task.",
      "reporting_status": "reported",
      "root_family_id": "atom3d",
      "task_type_id": "protein-complex-mutation-stability-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": "2026-07-21",
      "audit_status": "audited-with-caveats",
      "benchmark_id": "cameo",
      "benchmark_kind": "competition",
      "benchmark_name": "CAMEO",
      "benchmark_version": "current-complex-3d",
      "classification_status": "partial",
      "confidence": "high",
      "count": null,
      "count_basis": "Weekly complete-complex targets in the current CAMEO service.",
      "count_ref": null,
      "count_unit": "targets",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "cameo-paper"
      ],
      "evaluation_run_ids": [
        "cameo-2024-antibody-three-server-common",
        "cameo-2024-ligand-baseline-common",
        "cameo-2024-ppi-three-server-common"
      ],
      "evidence_ids": [
        "cameo-evidence-current-help"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "No fixed all-time count exists for the rolling service.",
      "reporting_status": "not_reported",
      "root_family_id": "cameo",
      "task_type_id": "protein-complex-structure-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": "2026-07-21",
      "audit_status": "audited-with-caveats",
      "benchmark_id": "casp",
      "benchmark_kind": "competition",
      "benchmark_name": "CASP",
      "benchmark_version": "CASP17",
      "classification_status": "partial",
      "confidence": "high",
      "count": 61,
      "count_basis": "CASP17 multimer target re-releases with stoichiometry as of 2026-07-21.",
      "count_ref": "/coverage_notes/0/count",
      "count_unit": "targets",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "casp-evidence-casp17-counts"
      ],
      "mapping_method": "official-track",
      "notes": "Rolling release count; not a final assessed-set total.",
      "reporting_status": "reported",
      "root_family_id": "casp",
      "task_type_id": "protein-complex-structure-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": "2026-07-21",
      "audit_status": "audited",
      "benchmark_id": "casp-protein-multimers",
      "benchmark_kind": "track",
      "benchmark_name": "CASP Protein Multimers",
      "benchmark_version": "CASP17",
      "classification_status": "complete",
      "confidence": "high",
      "count": 61,
      "count_basis": "CASP17 multimer target re-releases with stoichiometry as of 2026-07-21.",
      "count_ref": "/task_counts/total",
      "count_unit": "targets",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "casp16-multimer-assessment"
      ],
      "evaluation_run_ids": [
        "casp16-multimer-phase1-regular"
      ],
      "evidence_ids": [
        "casp-multimer-evidence-casp17"
      ],
      "mapping_method": "official-track",
      "notes": "Rolling release count; not a final assessed-set total.",
      "reporting_status": "reported",
      "root_family_id": "casp",
      "task_type_id": "protein-complex-structure-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "tape",
      "benchmark_kind": "suite",
      "benchmark_name": "TAPE",
      "benchmark_version": "original-2019",
      "classification_status": "complete",
      "confidence": "high",
      "count": 1,
      "count_basis": "Supervised downstream benchmark tasks.",
      "count_ref": null,
      "count_unit": "tasks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "tape-paper"
      ],
      "evaluation_run_ids": [
        "tape-creator-full"
      ],
      "evidence_ids": [
        "tape-paper-definition-evidence"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "Residue-pair contact prediction is the second structure task; it is not relabeled as full 3D folding.",
      "reporting_status": "reported",
      "root_family_id": "tape",
      "task_type_id": "protein-contact-map-prediction"
    },
    {
      "aggregate_eligible": false,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lifescibench",
      "benchmark_kind": "agentic-eval",
      "benchmark_name": "LifeSciBench",
      "benchmark_version": "initial-release",
      "classification_status": "partial",
      "confidence": "high",
      "count": 62,
      "count_basis": "Expert-authored tasks in the Protein primary domain and Design / Optimization workflow cell.",
      "count_ref": "/coverage_notes/1/count",
      "count_unit": "tasks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "lifescibench-preprint"
      ],
      "evaluation_run_ids": [
        "lifescibench-official-full"
      ],
      "evidence_ids": [
        "lifescibench-evidence-counts"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "This is a broad design-or-optimization count; it is not relabeled as sequence generation.",
      "reporting_status": "reported",
      "root_family_id": "lifescibench",
      "task_type_id": "protein-design"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "proteinlmbench",
      "benchmark_kind": "dataset",
      "benchmark_name": "ProteinLMBench",
      "benchmark_version": "hf-f139796",
      "classification_status": "partial",
      "confidence": "high",
      "count": null,
      "count_basis": "Released ProteinLMBench question records.",
      "count_ref": null,
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "proteinlmbench-paper"
      ],
      "evaluation_run_ids": [
        "proteinlmbench-creator-full"
      ],
      "evidence_ids": [
        "proteinlmbench-evidence-paper"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "Broad protein-design topic; no sequence-generation count is available.",
      "reporting_status": "not_reported",
      "root_family_id": "proteinlmbench",
      "task_type_id": "protein-design"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "flip",
      "benchmark_kind": "suite",
      "benchmark_name": "FLIP",
      "benchmark_version": "original-2021",
      "classification_status": "complete",
      "confidence": "high",
      "count": 15,
      "count_basis": "Dataset-and-split benchmark tasks.",
      "count_ref": "/task_counts/total",
      "count_unit": "tasks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "flip-evidence-paper-definition"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "All formal tasks are supervised fitness or phenotype prediction.",
      "reporting_status": "reported",
      "root_family_id": "flip",
      "task_type_id": "protein-fitness-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "flip-aav",
      "benchmark_kind": "track",
      "benchmark_name": "FLIP AAV",
      "benchmark_version": "original-2021",
      "classification_status": "complete",
      "confidence": "high",
      "count": 284009,
      "count_basis": "Distinct sequence-fitness examples across the sampled and designed AAV pools.",
      "count_ref": "/task_counts/total",
      "count_unit": "examples",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "flip-paper"
      ],
      "evaluation_run_ids": [
        "flip-aav-des-mut",
        "flip-aav-low-vs-high",
        "flip-aav-mut-des",
        "flip-aav-one-vs-rest",
        "flip-aav-sampled",
        "flip-aav-seven-vs-rest",
        "flip-aav-two-vs-rest"
      ],
      "evidence_ids": [
        "flip-aav-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Example count, not a count of formal split tasks.",
      "reporting_status": "reported",
      "root_family_id": "flip",
      "task_type_id": "protein-fitness-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "flip-gb1",
      "benchmark_kind": "track",
      "benchmark_name": "FLIP GB1",
      "benchmark_version": "original-2021",
      "classification_status": "complete",
      "confidence": "high",
      "count": 8733,
      "count_basis": "Downsampled sequence-fitness examples retained for FLIP.",
      "count_ref": "/task_counts/total",
      "count_unit": "examples",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "flip-paper"
      ],
      "evaluation_run_ids": [
        "flip-gb1-low-vs-high",
        "flip-gb1-one-vs-rest",
        "flip-gb1-sampled",
        "flip-gb1-three-vs-rest",
        "flip-gb1-two-vs-rest"
      ],
      "evidence_ids": [
        "flip-gb1-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Binding context does not turn the task into direct affinity prediction.",
      "reporting_status": "reported",
      "root_family_id": "flip",
      "task_type_id": "protein-fitness-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "proteingym",
      "benchmark_kind": "suite",
      "benchmark_name": "ProteinGym",
      "benchmark_version": "1.3",
      "classification_status": "partial",
      "confidence": "high",
      "count": null,
      "count_basis": "ProteinGym DMS assays.",
      "count_ref": null,
      "count_unit": "assays",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "system-card-claude-opus-5"
      ],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "proteingym-evidence-taxonomy"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "No heterogeneous root total is asserted.",
      "reporting_status": "not_reported",
      "root_family_id": "proteingym",
      "task_type_id": "protein-fitness-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "proteingym-dms-indels",
      "benchmark_kind": "track",
      "benchmark_name": "ProteinGym DMS Indels",
      "benchmark_version": "1.3",
      "classification_status": "complete",
      "confidence": "high",
      "count": 66,
      "count_basis": "DMS assays.",
      "count_ref": "/task_counts/total",
      "count_unit": "assays",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "proteingym-dms-indel-evidence-v13"
      ],
      "mapping_method": "official-track",
      "notes": "Overlapping claim; not an additive task total.",
      "reporting_status": "reported",
      "root_family_id": "proteingym",
      "task_type_id": "protein-fitness-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "proteingym-dms-substitutions",
      "benchmark_kind": "track",
      "benchmark_name": "ProteinGym DMS Substitutions",
      "benchmark_version": "1.3",
      "classification_status": "complete",
      "confidence": "high",
      "count": 217,
      "count_basis": "DMS assays.",
      "count_ref": "/task_counts/total",
      "count_unit": "assays",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "proteingym-paper"
      ],
      "evaluation_run_ids": [
        "proteingym-v10-dms-substitutions-zero-shot"
      ],
      "evidence_ids": [
        "proteingym-dms-sub-evidence-v13"
      ],
      "mapping_method": "official-track",
      "notes": "Overlapping scientific-task claim; never summed with mutation-effect coverage.",
      "reporting_status": "reported",
      "root_family_id": "proteingym",
      "task_type_id": "protein-fitness-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "bioinstruction",
      "benchmark_kind": "suite",
      "benchmark_name": "Biology-Instructions",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 1,
      "count_basis": "Formal evaluation tracks (Protein Fluorescence Prediction).",
      "count_ref": null,
      "count_unit": "tracks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "bioinstruction-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Track count; not an example count.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "protein-fluorescence-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "bioinstruction-fluorescence",
      "benchmark_kind": "track",
      "benchmark_name": "Biology-Instructions Protein Fluorescence Prediction",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 54025,
      "count_basis": "distinct examples across the published train, validation, and test splits",
      "count_ref": "/task_counts/total",
      "count_unit": "examples",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-fluorescence-closed-baselines",
        "bioinstruction-fluorescence-creator-systems",
        "bioinstruction-fluorescence-open-baselines"
      ],
      "evidence_ids": [
        "bioinstruction-fluorescence-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Protein fluorescence regression.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "protein-fluorescence-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "tape",
      "benchmark_kind": "suite",
      "benchmark_name": "TAPE",
      "benchmark_version": "original-2019",
      "classification_status": "complete",
      "confidence": "high",
      "count": 1,
      "count_basis": "Supervised downstream benchmark tasks.",
      "count_ref": null,
      "count_unit": "tasks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "tape-paper"
      ],
      "evaluation_run_ids": [
        "tape-creator-full"
      ],
      "evidence_ids": [
        "tape-paper-definition-evidence"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "Sequence-to-fluorescence regression; this is prediction, not sequence generation.",
      "reporting_status": "reported",
      "root_family_id": "tape",
      "task_type_id": "protein-fluorescence-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "bioinstruction",
      "benchmark_kind": "suite",
      "benchmark_name": "Biology-Instructions",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 1,
      "count_basis": "Formal evaluation tracks (Enzyme Commission Number Prediction).",
      "count_ref": null,
      "count_unit": "tracks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "bioinstruction-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Track count; not an example count.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "protein-function-annotation"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "bioinstruction-ec",
      "benchmark_kind": "track",
      "benchmark_name": "Biology-Instructions Enzyme Commission Number Prediction",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 19199,
      "count_basis": "distinct examples across the published train, validation, and test splits",
      "count_ref": "/task_counts/total",
      "count_unit": "examples",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-ec-closed-baselines",
        "bioinstruction-ec-creator-systems",
        "bioinstruction-ec-open-baselines"
      ],
      "evidence_ids": [
        "bioinstruction-ec-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Enzyme Commission function annotation.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "protein-function-annotation"
    },
    {
      "aggregate_eligible": true,
      "as_of": "2026-07-22",
      "audit_status": "audited",
      "benchmark_id": "atom3d",
      "benchmark_kind": "suite",
      "benchmark_name": "ATOM3D",
      "benchmark_version": "v0.2.6",
      "classification_status": "complete",
      "confidence": "high",
      "count": 1,
      "count_basis": "Curated 3D benchmark datasets.",
      "count_ref": null,
      "count_unit": "tasks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "atom3d-paper"
      ],
      "evaluation_run_ids": [
        "atom3d-creator-full"
      ],
      "evidence_ids": [
        "atom3d-paper-definition-evidence"
      ],
      "mapping_method": "official-track",
      "notes": "LBA, ligand binding-affinity prediction.",
      "reporting_status": "reported",
      "root_family_id": "atom3d",
      "task_type_id": "protein-ligand-binding-affinity"
    },
    {
      "aggregate_eligible": true,
      "as_of": "2026-07-21",
      "audit_status": "audited-with-caveats",
      "benchmark_id": "casp",
      "benchmark_kind": "competition",
      "benchmark_name": "CASP",
      "benchmark_version": "CASP17",
      "classification_status": "partial",
      "confidence": "high",
      "count": null,
      "count_basis": "CASP17 protein-ligand targets.",
      "count_ref": null,
      "count_unit": "targets",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "casp-evidence-casp17-protocol"
      ],
      "mapping_method": "official-track",
      "notes": "Affinity or rank prediction is a formal ligand-category objective.",
      "reporting_status": "not_reported",
      "root_family_id": "casp",
      "task_type_id": "protein-ligand-binding-affinity"
    },
    {
      "aggregate_eligible": true,
      "as_of": "2026-07-21",
      "audit_status": "audited",
      "benchmark_id": "casp-protein-ligands",
      "benchmark_kind": "track",
      "benchmark_name": "CASP Protein-Ligand Prediction",
      "benchmark_version": "CASP17",
      "classification_status": "complete",
      "confidence": "high",
      "count": null,
      "count_basis": "CASP17 protein-ligand targets.",
      "count_ref": null,
      "count_unit": "targets",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "casp16-ligand-assessment"
      ],
      "evaluation_run_ids": [
        "casp16-ligand-affinity-stage1",
        "casp16-ligand-affinity-stage2",
        "casp16-ligand-pose-regular"
      ],
      "evidence_ids": [
        "casp-ligand-evidence-casp17"
      ],
      "mapping_method": "official-track",
      "notes": "Affinity or ranking is explicit, without a current standalone count.",
      "reporting_status": "not_reported",
      "root_family_id": "casp",
      "task_type_id": "protein-ligand-binding-affinity"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "moleculenet",
      "benchmark_kind": "suite",
      "benchmark_name": "MoleculeNet",
      "benchmark_version": "original-2017",
      "classification_status": "partial",
      "confidence": "high",
      "count": 1,
      "count_basis": "Original paper dataset collections.",
      "count_ref": null,
      "count_unit": "other",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "enhancing-molecular-property-prediction-with-auxiliary",
        "moleculenet-paper"
      ],
      "evaluation_run_ids": [
        "moleculenet-creator-full"
      ],
      "evidence_ids": [
        "moleculenet-paper-definition-evidence"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "PDBbind is the explicitly structure-based protein-ligand affinity collection.",
      "reporting_status": "reported",
      "root_family_id": "moleculenet",
      "task_type_id": "protein-ligand-binding-affinity"
    },
    {
      "aggregate_eligible": false,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lifescibench",
      "benchmark_kind": "agentic-eval",
      "benchmark_name": "LifeSciBench",
      "benchmark_version": "initial-release",
      "classification_status": "partial",
      "confidence": "high",
      "count": null,
      "count_basis": "Expert-authored benchmark tasks.",
      "count_ref": null,
      "count_unit": "tasks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "lifescibench-preprint"
      ],
      "evaluation_run_ids": [
        "lifescibench-official-full"
      ],
      "evidence_ids": [
        "lifescibench-evidence-taxonomy"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "The source does not distinguish pose from affinity, so the broad task is retained.",
      "reporting_status": "not_reported",
      "root_family_id": "lifescibench",
      "task_type_id": "protein-ligand-binding-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "proteingym",
      "benchmark_kind": "suite",
      "benchmark_name": "ProteinGym",
      "benchmark_version": "1.3",
      "classification_status": "partial",
      "confidence": "high",
      "count": null,
      "count_basis": "DMS assays in the official generic Binding function category.",
      "count_ref": null,
      "count_unit": "assays",
      "coverage": "observed",
      "evaluating_work_ids": [
        "system-card-claude-opus-5"
      ],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "proteingym-evidence-taxonomy"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "The source does not support a ligand-only or PPI-only assay count.",
      "reporting_status": "not_reported",
      "root_family_id": "proteingym",
      "task_type_id": "protein-ligand-binding-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": "2026-07-22",
      "audit_status": "audited",
      "benchmark_id": "atom3d",
      "benchmark_kind": "suite",
      "benchmark_name": "ATOM3D",
      "benchmark_version": "v0.2.6",
      "classification_status": "complete",
      "confidence": "high",
      "count": 1,
      "count_basis": "Curated 3D benchmark datasets.",
      "count_ref": null,
      "count_unit": "tasks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "atom3d-paper"
      ],
      "evaluation_run_ids": [
        "atom3d-creator-full"
      ],
      "evidence_ids": [
        "atom3d-paper-definition-evidence"
      ],
      "mapping_method": "official-track",
      "notes": "LEP classifies whether a bound molecule activates protein function from active and inactive target conformations; it does not predict binding presence or affinity.",
      "reporting_status": "reported",
      "root_family_id": "atom3d",
      "task_type_id": "protein-ligand-efficacy-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": "2026-07-21",
      "audit_status": "audited-with-caveats",
      "benchmark_id": "cameo",
      "benchmark_kind": "competition",
      "benchmark_name": "CAMEO",
      "benchmark_version": "current-complex-3d",
      "classification_status": "partial",
      "confidence": "high",
      "count": null,
      "count_basis": "Ligand-containing targets in the current rolling service.",
      "count_ref": null,
      "count_unit": "targets",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "cameo-paper"
      ],
      "evaluation_run_ids": [
        "cameo-2024-antibody-three-server-common",
        "cameo-2024-ligand-baseline-common",
        "cameo-2024-ppi-three-server-common"
      ],
      "evidence_ids": [
        "cameo-evidence-2024-study"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "Ligand pose is in scope; binding affinity is not asserted.",
      "reporting_status": "not_reported",
      "root_family_id": "cameo",
      "task_type_id": "protein-ligand-pose-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": "2026-07-21",
      "audit_status": "audited-with-caveats",
      "benchmark_id": "casp",
      "benchmark_kind": "competition",
      "benchmark_name": "CASP",
      "benchmark_version": "CASP17",
      "classification_status": "partial",
      "confidence": "high",
      "count": null,
      "count_basis": "CASP17 protein-ligand targets.",
      "count_ref": null,
      "count_unit": "targets",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "casp-evidence-casp17-protocol"
      ],
      "mapping_method": "official-track",
      "notes": "Pose prediction is a formal ligand-category objective.",
      "reporting_status": "not_reported",
      "root_family_id": "casp",
      "task_type_id": "protein-ligand-pose-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": "2026-07-21",
      "audit_status": "audited",
      "benchmark_id": "casp-protein-ligands",
      "benchmark_kind": "track",
      "benchmark_name": "CASP Protein-Ligand Prediction",
      "benchmark_version": "CASP17",
      "classification_status": "complete",
      "confidence": "high",
      "count": null,
      "count_basis": "CASP17 protein-ligand targets.",
      "count_ref": null,
      "count_unit": "targets",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "casp16-ligand-assessment"
      ],
      "evaluation_run_ids": [
        "casp16-ligand-affinity-stage1",
        "casp16-ligand-affinity-stage2",
        "casp16-ligand-pose-regular"
      ],
      "evidence_ids": [
        "casp-ligand-evidence-casp17"
      ],
      "mapping_method": "official-track",
      "notes": "Pose and pocket prediction are explicit objectives.",
      "reporting_status": "not_reported",
      "root_family_id": "casp",
      "task_type_id": "protein-ligand-pose-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": "2026-07-22",
      "audit_status": "audited",
      "benchmark_id": "atom3d",
      "benchmark_kind": "suite",
      "benchmark_name": "ATOM3D",
      "benchmark_version": "v0.2.6",
      "classification_status": "complete",
      "confidence": "high",
      "count": 1,
      "count_basis": "Curated 3D benchmark datasets.",
      "count_ref": null,
      "count_unit": "tasks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "atom3d-paper"
      ],
      "evaluation_run_ids": [
        "atom3d-creator-full"
      ],
      "evidence_ids": [
        "atom3d-paper-definition-evidence"
      ],
      "mapping_method": "official-track",
      "notes": "PSR, protein structure ranking.",
      "reporting_status": "reported",
      "root_family_id": "atom3d",
      "task_type_id": "protein-model-quality-assessment"
    },
    {
      "aggregate_eligible": true,
      "as_of": "2026-07-21",
      "audit_status": "audited-with-caveats",
      "benchmark_id": "cameo",
      "benchmark_kind": "competition",
      "benchmark_name": "CAMEO",
      "benchmark_version": "current-complex-3d",
      "classification_status": "partial",
      "confidence": "high",
      "count": null,
      "count_basis": "Weekly targets and submitted server models.",
      "count_ref": null,
      "count_unit": "targets",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "cameo-paper"
      ],
      "evaluation_run_ids": [
        "cameo-2024-antibody-three-server-common",
        "cameo-2024-ligand-baseline-common",
        "cameo-2024-ppi-three-server-common"
      ],
      "evidence_ids": [
        "cameo-evidence-2024-study"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "CAMEO evaluates prediction quality against newly released structures.",
      "reporting_status": "not_reported",
      "root_family_id": "cameo",
      "task_type_id": "protein-model-quality-assessment"
    },
    {
      "aggregate_eligible": true,
      "as_of": "2026-07-21",
      "audit_status": "audited-with-caveats",
      "benchmark_id": "casp",
      "benchmark_kind": "competition",
      "benchmark_name": "CASP",
      "benchmark_version": "CASP17",
      "classification_status": "partial",
      "confidence": "high",
      "count": null,
      "count_basis": "CASP17 model-selection and accuracy-estimation targets.",
      "count_ref": null,
      "count_unit": "targets",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "casp-evidence-casp17-protocol"
      ],
      "mapping_method": "official-track",
      "notes": "No final current-round count is asserted.",
      "reporting_status": "not_reported",
      "root_family_id": "casp",
      "task_type_id": "protein-model-quality-assessment"
    },
    {
      "aggregate_eligible": true,
      "as_of": "2026-07-21",
      "audit_status": "audited",
      "benchmark_id": "casp-protein-multimers",
      "benchmark_kind": "track",
      "benchmark_name": "CASP Protein Multimers",
      "benchmark_version": "CASP17",
      "classification_status": "complete",
      "confidence": "high",
      "count": null,
      "count_basis": "CASP17 multimer model-selection evaluation units.",
      "count_ref": null,
      "count_unit": "targets",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "casp16-multimer-assessment"
      ],
      "evaluation_run_ids": [
        "casp16-multimer-phase1-regular"
      ],
      "evidence_ids": [
        "casp-multimer-evidence-casp17-protocol"
      ],
      "mapping_method": "official-track",
      "notes": "Model-selection is in scope but not separately counted.",
      "reporting_status": "not_reported",
      "root_family_id": "casp",
      "task_type_id": "protein-model-quality-assessment"
    },
    {
      "aggregate_eligible": true,
      "as_of": "2026-07-21",
      "audit_status": "audited-with-caveats",
      "benchmark_id": "casp",
      "benchmark_kind": "competition",
      "benchmark_name": "CASP",
      "benchmark_version": "CASP17",
      "classification_status": "partial",
      "confidence": "high",
      "count": null,
      "count_basis": "CASP17 assessed monomer evaluation units.",
      "count_ref": null,
      "count_unit": "targets",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "casp-evidence-casp17-protocol"
      ],
      "mapping_method": "official-track",
      "notes": "The final assessed monomer count is not yet reported.",
      "reporting_status": "not_reported",
      "root_family_id": "casp",
      "task_type_id": "protein-monomer-structure-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": "2026-07-21",
      "audit_status": "audited-with-caveats",
      "benchmark_id": "casp-protein-monomers",
      "benchmark_kind": "track",
      "benchmark_name": "CASP Protein Monomers",
      "benchmark_version": "CASP17",
      "classification_status": "complete",
      "confidence": "high",
      "count": null,
      "count_basis": "CASP17 assessed monomer evaluation units.",
      "count_ref": null,
      "count_unit": "targets",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "casp16-monomer-assessment"
      ],
      "evaluation_run_ids": [
        "casp16-monomer-regular-official"
      ],
      "evidence_ids": [
        "casp-monomer-evidence-casp17"
      ],
      "mapping_method": "official-track",
      "notes": "Final assessed count is not reported while CASP17 is active.",
      "reporting_status": "not_reported",
      "root_family_id": "casp",
      "task_type_id": "protein-monomer-structure-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "flip",
      "benchmark_kind": "suite",
      "benchmark_name": "FLIP",
      "benchmark_version": "original-2021",
      "classification_status": "complete",
      "confidence": "high",
      "count": 15,
      "count_basis": "Dataset-and-split benchmark tasks.",
      "count_ref": "/task_counts/total",
      "count_unit": "tasks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "flip-evidence-paper-definition"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "The same tasks evaluate generalization across mutated sequence landscapes; claims overlap and are never summed.",
      "reporting_status": "reported",
      "root_family_id": "flip",
      "task_type_id": "protein-mutation-effect-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "proteingym",
      "benchmark_kind": "suite",
      "benchmark_name": "ProteinGym",
      "benchmark_version": "1.3",
      "classification_status": "partial",
      "confidence": "high",
      "count": null,
      "count_basis": "DMS assays and mutant measurements across version 1.3 tracks.",
      "count_ref": null,
      "count_unit": "assays",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "system-card-claude-opus-5"
      ],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "proteingym-evidence-taxonomy"
      ],
      "mapping_method": "official-track",
      "notes": "Track-specific assay counts are recorded on child records.",
      "reporting_status": "not_reported",
      "root_family_id": "proteingym",
      "task_type_id": "protein-mutation-effect-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "proteingym-dms-indels",
      "benchmark_kind": "track",
      "benchmark_name": "ProteinGym DMS Indels",
      "benchmark_version": "1.3",
      "classification_status": "complete",
      "confidence": "high",
      "count": 66,
      "count_basis": "DMS assays.",
      "count_ref": "/task_counts/total",
      "count_unit": "assays",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "proteingym-dms-indel-evidence-v13"
      ],
      "mapping_method": "official-track",
      "notes": "The underlying benchmark field carries the audit caveat.",
      "reporting_status": "reported",
      "root_family_id": "proteingym",
      "task_type_id": "protein-mutation-effect-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "proteingym-dms-substitutions",
      "benchmark_kind": "track",
      "benchmark_name": "ProteinGym DMS Substitutions",
      "benchmark_version": "1.3",
      "classification_status": "complete",
      "confidence": "high",
      "count": 217,
      "count_basis": "DMS assays.",
      "count_ref": "/task_counts/total",
      "count_unit": "assays",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "proteingym-paper"
      ],
      "evaluation_run_ids": [
        "proteingym-v10-dms-substitutions-zero-shot"
      ],
      "evidence_ids": [
        "proteingym-dms-sub-evidence-v13"
      ],
      "mapping_method": "official-track",
      "notes": "Assay count; not mutant-record count.",
      "reporting_status": "reported",
      "root_family_id": "proteingym",
      "task_type_id": "protein-mutation-effect-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "lab-bench",
      "benchmark_kind": "suite",
      "benchmark_name": "LAB-Bench",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "partial",
      "confidence": "high",
      "count": 50,
      "count_basis": "Viral PPI formal-task questions.",
      "count_ref": "/coverage_notes/1/count",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "lab-bench-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Viral-human PPI database-retrieval questions.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "protein-protein-interaction-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-dbqa-viral-ppi",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench DbQA — Viral protein–protein interactions",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "complete",
      "confidence": "high",
      "count": 50,
      "count_basis": "questions across public and private splits",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-dbqa-viral-ppi-creator-mcq"
      ],
      "evidence_ids": [
        "lab-bench-dbqa-viral-ppi-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "The official task retrieves predicted P-HIPSter interaction partners.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "protein-protein-interaction-prediction"
    },
    {
      "aggregate_eligible": false,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lifescibench",
      "benchmark_kind": "agentic-eval",
      "benchmark_name": "LifeSciBench",
      "benchmark_version": "initial-release",
      "classification_status": "partial",
      "confidence": "high",
      "count": null,
      "count_basis": "Expert-authored benchmark tasks.",
      "count_ref": null,
      "count_unit": "tasks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "lifescibench-preprint"
      ],
      "evaluation_run_ids": [
        "lifescibench-official-full"
      ],
      "evidence_ids": [
        "lifescibench-evidence-taxonomy"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "The official protein-domain definition and examples include protein-protein binding, without a standalone count.",
      "reporting_status": "not_reported",
      "root_family_id": "lifescibench",
      "task_type_id": "protein-protein-interaction-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": "2026-07-22",
      "audit_status": "audited",
      "benchmark_id": "atom3d",
      "benchmark_kind": "suite",
      "benchmark_name": "ATOM3D",
      "benchmark_version": "v0.2.6",
      "classification_status": "complete",
      "confidence": "high",
      "count": 1,
      "count_basis": "Curated 3D benchmark datasets.",
      "count_ref": null,
      "count_unit": "tasks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "atom3d-paper"
      ],
      "evaluation_run_ids": [
        "atom3d-creator-full"
      ],
      "evidence_ids": [
        "atom3d-paper-definition-evidence"
      ],
      "mapping_method": "official-track",
      "notes": "PIP predicts whether residue pairs from two proteins contact when the proteins bind; it is narrower than binary interaction detection.",
      "reporting_status": "reported",
      "root_family_id": "atom3d",
      "task_type_id": "protein-protein-interface-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "tape",
      "benchmark_kind": "suite",
      "benchmark_name": "TAPE",
      "benchmark_version": "original-2019",
      "classification_status": "complete",
      "confidence": "high",
      "count": 1,
      "count_basis": "Supervised downstream benchmark tasks.",
      "count_ref": null,
      "count_unit": "tasks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "tape-paper"
      ],
      "evaluation_run_ids": [
        "tape-creator-full"
      ],
      "evidence_ids": [
        "tape-paper-definition-evidence"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "Fold-level remote-homology classification.",
      "reporting_status": "reported",
      "root_family_id": "tape",
      "task_type_id": "protein-remote-homology-detection"
    },
    {
      "aggregate_eligible": true,
      "as_of": "2026-07-22",
      "audit_status": "audited",
      "benchmark_id": "atom3d",
      "benchmark_kind": "suite",
      "benchmark_name": "ATOM3D",
      "benchmark_version": "v0.2.6",
      "classification_status": "complete",
      "confidence": "high",
      "count": 1,
      "count_basis": "Curated 3D benchmark datasets.",
      "count_ref": null,
      "count_unit": "tasks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "atom3d-paper"
      ],
      "evaluation_run_ids": [
        "atom3d-creator-full"
      ],
      "evidence_ids": [
        "atom3d-paper-definition-evidence"
      ],
      "mapping_method": "official-track",
      "notes": "RES, residue identity prediction from the local structural environment.",
      "reporting_status": "reported",
      "root_family_id": "atom3d",
      "task_type_id": "protein-residue-identity-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "tape",
      "benchmark_kind": "suite",
      "benchmark_name": "TAPE",
      "benchmark_version": "original-2019",
      "classification_status": "complete",
      "confidence": "high",
      "count": 1,
      "count_basis": "Supervised downstream benchmark tasks.",
      "count_ref": null,
      "count_unit": "tasks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "tape-paper"
      ],
      "evaluation_run_ids": [
        "tape-creator-full"
      ],
      "evidence_ids": [
        "tape-paper-definition-evidence"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "Per-residue three-state and eight-state secondary-structure prediction constitute one benchmark task.",
      "reporting_status": "reported",
      "root_family_id": "tape",
      "task_type_id": "protein-secondary-structure-prediction"
    },
    {
      "aggregate_eligible": false,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "bioinstruction",
      "benchmark_kind": "suite",
      "benchmark_name": "Biology-Instructions",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 0,
      "count_basis": "Formal evaluation tracks in the final creator paper.",
      "count_ref": "/coverage_notes/3/count",
      "count_unit": "tracks",
      "coverage": "not-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "bioinstruction-evidence-paper"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "The paper explicitly leaves generative sequence design for future work.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "protein-sequence-design"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "cam-benchmark",
      "benchmark_kind": "dataset",
      "benchmark_name": "CaM benchmark",
      "benchmark_version": "initial-release",
      "classification_status": "partial",
      "confidence": "high",
      "count": null,
      "count_basis": "Creator-defined multistate protein sequence-design benchmark system.",
      "count_ref": null,
      "count_unit": "systems",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "an-integrative-approach-to-protein-sequence-design-thr"
      ],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "cam-benchmark-automated-metadata-2-evidence"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "The fourteen modeled structural states are objective dimensions, not fourteen independent sequence-design tasks.",
      "reporting_status": "not_reported",
      "root_family_id": "cam-benchmark",
      "task_type_id": "protein-sequence-design"
    },
    {
      "aggregate_eligible": false,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "lab-bench",
      "benchmark_kind": "suite",
      "benchmark_name": "LAB-Bench",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "partial",
      "confidence": "high",
      "count": 0,
      "count_basis": "Released LAB-Bench questions.",
      "count_ref": "/coverage_notes/2/count",
      "count_unit": "questions",
      "coverage": "not-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "lab-bench-evidence-paper"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "Primer and cloning design are not relabeled as protein sequence design.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "protein-sequence-design"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "papd-benchmark",
      "benchmark_kind": "dataset",
      "benchmark_name": "PapD benchmark",
      "benchmark_version": "initial-release",
      "classification_status": "partial",
      "confidence": "high",
      "count": null,
      "count_basis": "Creator-defined multistate protein sequence-design benchmark system.",
      "count_ref": null,
      "count_unit": "systems",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "an-integrative-approach-to-protein-sequence-design-thr"
      ],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "papd-benchmark-automated-metadata-2-evidence"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "The three modeled binding states are objective dimensions, not three independent sequence-design tasks.",
      "reporting_status": "not_reported",
      "root_family_id": "papd-benchmark",
      "task_type_id": "protein-sequence-design"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "rfah-benchmark",
      "benchmark_kind": "dataset",
      "benchmark_name": "RfaH benchmark",
      "benchmark_version": "initial-release",
      "classification_status": "partial",
      "confidence": "high",
      "count": null,
      "count_basis": "Creator-defined multistate protein sequence-design benchmark system.",
      "count_ref": null,
      "count_unit": "systems",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "an-integrative-approach-to-protein-sequence-design-thr"
      ],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "rfah-benchmark-automated-metadata-2-evidence"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "The two modeled conformational states are objective dimensions, not two independent sequence-design tasks.",
      "reporting_status": "not_reported",
      "root_family_id": "rfah-benchmark",
      "task_type_id": "protein-sequence-design"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "bioinstruction",
      "benchmark_kind": "suite",
      "benchmark_name": "Biology-Instructions",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 1,
      "count_basis": "Formal evaluation tracks (Protein Solubility Prediction).",
      "count_ref": null,
      "count_unit": "tracks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "bioinstruction-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Track count; not an example count.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "protein-solubility-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "bioinstruction-solubility",
      "benchmark_kind": "track",
      "benchmark_name": "Biology-Instructions Protein Solubility Prediction",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 71421,
      "count_basis": "distinct examples across the published train, validation, and test splits",
      "count_ref": "/task_counts/total",
      "count_unit": "examples",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-solubility-closed-baselines",
        "bioinstruction-solubility-creator-systems",
        "bioinstruction-solubility-open-baselines"
      ],
      "evidence_ids": [
        "bioinstruction-solubility-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Protein solubility prediction.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "protein-solubility-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "bioinstruction",
      "benchmark_kind": "suite",
      "benchmark_name": "Biology-Instructions",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 2,
      "count_basis": "Formal evaluation tracks (Protein Stability and Protein Thermostability).",
      "count_ref": null,
      "count_unit": "tracks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "bioinstruction-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Track count; not an example count.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "protein-stability-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "bioinstruction-stability",
      "benchmark_kind": "track",
      "benchmark_name": "Biology-Instructions Protein Stability Prediction",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 68977,
      "count_basis": "distinct examples across the published train, validation, and test splits",
      "count_ref": "/task_counts/total",
      "count_unit": "examples",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-stability-closed-baselines",
        "bioinstruction-stability-creator-systems",
        "bioinstruction-stability-open-baselines"
      ],
      "evidence_ids": [
        "bioinstruction-stability-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Protein stability prediction.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "protein-stability-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "bioinstruction-thermostability",
      "benchmark_kind": "track",
      "benchmark_name": "Biology-Instructions Protein Thermostability Prediction",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 7031,
      "count_basis": "distinct examples across the published train, validation, and test splits",
      "count_ref": "/task_counts/total",
      "count_unit": "examples",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-thermostability-closed-baselines",
        "bioinstruction-thermostability-creator-systems",
        "bioinstruction-thermostability-open-baselines"
      ],
      "evidence_ids": [
        "bioinstruction-thermostability-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Protein thermostability prediction.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "protein-stability-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "flip-meltome",
      "benchmark_kind": "track",
      "benchmark_name": "FLIP Meltome Thermostability",
      "benchmark_version": "original-2021",
      "classification_status": "complete",
      "confidence": "high",
      "count": 27951,
      "count_basis": "Sequence-temperature examples in the Mixed split universe.",
      "count_ref": "/task_counts/total",
      "count_unit": "examples",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "flip-paper"
      ],
      "evaluation_run_ids": [
        "flip-meltome-human",
        "flip-meltome-human-cell",
        "flip-meltome-mixed"
      ],
      "evidence_ids": [
        "flip-meltome-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Temperature examples are not combined with AAV or GB1 counts.",
      "reporting_status": "reported",
      "root_family_id": "flip",
      "task_type_id": "protein-stability-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "tape",
      "benchmark_kind": "suite",
      "benchmark_name": "TAPE",
      "benchmark_version": "original-2019",
      "classification_status": "complete",
      "confidence": "high",
      "count": 1,
      "count_basis": "Supervised downstream benchmark tasks.",
      "count_ref": null,
      "count_unit": "tasks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "tape-paper"
      ],
      "evaluation_run_ids": [
        "tape-creator-full"
      ],
      "evidence_ids": [
        "tape-paper-definition-evidence"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "Sequence-to-stability regression; this is prediction, not sequence generation.",
      "reporting_status": "reported",
      "root_family_id": "tape",
      "task_type_id": "protein-stability-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "proteinlmbench",
      "benchmark_kind": "dataset",
      "benchmark_name": "ProteinLMBench",
      "benchmark_version": "hf-f139796",
      "classification_status": "partial",
      "confidence": "high",
      "count": null,
      "count_basis": "Released ProteinLMBench question records.",
      "count_ref": null,
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "proteinlmbench-paper"
      ],
      "evaluation_run_ids": [
        "proteinlmbench-creator-full"
      ],
      "evidence_ids": [
        "proteinlmbench-evidence-paper"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "Broad structure coverage only; folding is not inferred.",
      "reporting_status": "not_reported",
      "root_family_id": "proteinlmbench",
      "task_type_id": "protein-structure"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "scigym",
      "benchmark_kind": "suite",
      "benchmark_name": "SCIGYM",
      "benchmark_version": "2025 release",
      "classification_status": "complete",
      "confidence": "high",
      "count": 350,
      "count_basis": "distinct curated BioModels systems released as SBML benchmark instances",
      "count_ref": "/task_counts/total",
      "count_unit": "systems",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "scigym-evidence-release-counts",
        "scigym-evidence-taxonomy"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "Hidden biological reactions are reconstructed from interventions and trajectories.",
      "reporting_status": "reported",
      "root_family_id": "scigym",
      "task_type_id": "reaction-network-reconstruction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "scigym-large",
      "benchmark_kind": "track",
      "benchmark_name": "SCIGYM Large",
      "benchmark_version": "2025 release",
      "classification_status": "complete",
      "confidence": "high",
      "count": 213,
      "count_basis": "unique SBML systems in the official large Parquet split, containing the remaining systems with up to 400 reactions",
      "count_ref": "/task_counts/total",
      "count_unit": "systems",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "scigym-large-evidence-count"
      ],
      "mapping_method": "official-track",
      "notes": "Split-specific system count.",
      "reporting_status": "reported",
      "root_family_id": "scigym",
      "task_type_id": "reaction-network-reconstruction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "scigym-small",
      "benchmark_kind": "track",
      "benchmark_name": "SCIGYM Small",
      "benchmark_version": "2025 release",
      "classification_status": "complete",
      "confidence": "high",
      "count": 137,
      "count_basis": "unique SBML systems with fewer than 10 reactions in the official small Parquet split",
      "count_ref": "/task_counts/total",
      "count_unit": "systems",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "scigym-paper"
      ],
      "evaluation_run_ids": [
        "scigym-small-creator-paper",
        "scigym-small-zero-shot"
      ],
      "evidence_ids": [
        "scigym-small-evidence-count"
      ],
      "mapping_method": "official-track",
      "notes": "Split-specific system count.",
      "reporting_status": "reported",
      "root_family_id": "scigym",
      "task_type_id": "reaction-network-reconstruction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "bioinstruction",
      "benchmark_kind": "suite",
      "benchmark_name": "Biology-Instructions",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 1,
      "count_basis": "Formal evaluation tracks (Programmable RNA Switches).",
      "count_ref": null,
      "count_unit": "tracks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "bioinstruction-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Track count; not an example count.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "rna-design"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "bioinstruction-prs",
      "benchmark_kind": "track",
      "benchmark_name": "Biology-Instructions Programmable RNA Switches",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 93399,
      "count_basis": "distinct examples across the published train, validation, and test splits",
      "count_ref": "/task_counts/total",
      "count_unit": "examples",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-prs-closed-baselines",
        "bioinstruction-prs-creator-systems",
        "bioinstruction-prs-open-baselines"
      ],
      "evidence_ids": [
        "bioinstruction-prs-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Programmable RNA-switch value prediction.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "rna-design"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "beacon-rna",
      "benchmark_kind": "suite",
      "benchmark_name": "BEACON",
      "benchmark_version": "neurips-2024",
      "classification_status": "complete",
      "confidence": "high",
      "count": 1,
      "count_basis": "Formal RNA benchmark tasks.",
      "count_ref": null,
      "count_unit": "tasks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "beacon-paper"
      ],
      "evaluation_run_ids": [
        "beacon-creator-full"
      ],
      "evidence_ids": [
        "beacon-paper-definition-evidence"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "Non-coding RNA functional classification.",
      "reporting_status": "reported",
      "root_family_id": "beacon-rna",
      "task_type_id": "rna-function-classification"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "bioinstruction",
      "benchmark_kind": "suite",
      "benchmark_name": "Biology-Instructions",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 1,
      "count_basis": "Formal evaluation tracks (Non-coding RNA Function Classification).",
      "count_ref": null,
      "count_unit": "tracks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "bioinstruction-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Track count; not an example count.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "rna-function-classification"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "bioinstruction-ncrna",
      "benchmark_kind": "track",
      "benchmark_name": "Biology-Instructions Non-coding RNA Function Classification",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 11160,
      "count_basis": "distinct examples across the published train, validation, and test splits",
      "count_ref": "/task_counts/total",
      "count_unit": "examples",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-ncrna-closed-baselines",
        "bioinstruction-ncrna-creator-systems",
        "bioinstruction-ncrna-open-baselines"
      ],
      "evidence_ids": [
        "bioinstruction-ncrna-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Non-coding RNA function classification.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "rna-function-classification"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "beacon-rna",
      "benchmark_kind": "suite",
      "benchmark_name": "BEACON",
      "benchmark_version": "neurips-2024",
      "classification_status": "complete",
      "confidence": "high",
      "count": 1,
      "count_basis": "Formal RNA benchmark tasks.",
      "count_ref": null,
      "count_unit": "tasks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "beacon-paper"
      ],
      "evaluation_run_ids": [
        "beacon-creator-full"
      ],
      "evidence_ids": [
        "beacon-paper-definition-evidence"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "RNA modification-site prediction.",
      "reporting_status": "reported",
      "root_family_id": "beacon-rna",
      "task_type_id": "rna-modification-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "bioinstruction",
      "benchmark_kind": "suite",
      "benchmark_name": "Biology-Instructions",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 1,
      "count_basis": "Formal evaluation tracks (RNA Modification Prediction).",
      "count_ref": null,
      "count_unit": "tracks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "bioinstruction-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Track count; not an example count.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "rna-modification-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "bioinstruction-modification",
      "benchmark_kind": "track",
      "benchmark_name": "Biology-Instructions RNA Modification Prediction",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 309460,
      "count_basis": "distinct examples across the published train, validation, and test splits",
      "count_ref": "/task_counts/total",
      "count_unit": "examples",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-modification-closed-baselines",
        "bioinstruction-modification-creator-systems",
        "bioinstruction-modification-open-baselines"
      ],
      "evidence_ids": [
        "bioinstruction-modification-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "RNA chemical-modification prediction.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "rna-modification-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "beacon-rna",
      "benchmark_kind": "suite",
      "benchmark_name": "BEACON",
      "benchmark_version": "neurips-2024",
      "classification_status": "complete",
      "confidence": "high",
      "count": 3,
      "count_basis": "Formal RNA benchmark tasks.",
      "count_ref": null,
      "count_unit": "tasks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "beacon-paper"
      ],
      "evaluation_run_ids": [
        "beacon-creator-full"
      ],
      "evidence_ids": [
        "beacon-paper-definition-evidence"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "Splice-site prediction, alternative polyadenylation, and mean ribosome loading.",
      "reporting_status": "reported",
      "root_family_id": "beacon-rna",
      "task_type_id": "rna-processing-translation-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "bioinstruction",
      "benchmark_kind": "suite",
      "benchmark_name": "Biology-Instructions",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 2,
      "count_basis": "Formal evaluation tracks (APA Isoform Prediction and Mean Ribosome Loading).",
      "count_ref": null,
      "count_unit": "tracks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "bioinstruction-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Track count; not an example count.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "rna-processing-translation-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "bioinstruction-apa",
      "benchmark_kind": "track",
      "benchmark_name": "Biology-Instructions APA Isoform Prediction",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 1658482,
      "count_basis": "distinct examples across the published train, validation, and test splits",
      "count_ref": "/task_counts/total",
      "count_unit": "examples",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-apa-closed-baselines",
        "bioinstruction-apa-creator-systems",
        "bioinstruction-apa-open-baselines"
      ],
      "evidence_ids": [
        "bioinstruction-apa-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Alternative-polyadenylation isoform usage prediction.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "rna-processing-translation-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "bioinstruction-mrl",
      "benchmark_kind": "track",
      "benchmark_name": "Biology-Instructions Mean Ribosome Loading Prediction",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 91519,
      "count_basis": "distinct examples across the published train, validation, and test splits",
      "count_ref": "/task_counts/total",
      "count_unit": "examples",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-mrl-closed-baselines",
        "bioinstruction-mrl-creator-systems",
        "bioinstruction-mrl-open-baselines"
      ],
      "evidence_ids": [
        "bioinstruction-mrl-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Mean ribosome-loading prediction.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "rna-processing-translation-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-seqqa",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench SeqQA",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "partial",
      "confidence": "high",
      "count": 200,
      "count_basis": "Four 50-question ORF and translation formal child tracks.",
      "count_ref": null,
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "anthropic-sonnet-4-5-system-card"
      ],
      "evaluation_run_ids": [
        "lab-bench-seqqa-anthropic-sonnet45-system-card"
      ],
      "evidence_ids": [
        "lab-bench-seqqa-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "The count is derived only from formal child-track sizes documented by the official release.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "rna-processing-translation-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-seqqa-orf-seq-aaid",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench SeqQA — ORF amino-acid position",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "complete",
      "confidence": "high",
      "count": 50,
      "count_basis": "questions across public and private splits",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-seqqa-orf-seq-aaid-creator-mcq"
      ],
      "evidence_ids": [
        "lab-bench-seqqa-orf-seq-aaid-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Translation of a DNA open reading frame.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "rna-processing-translation-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-seqqa-orf-seq-aaseq",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench SeqQA — ORF amino-acid sequence",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "complete",
      "confidence": "high",
      "count": 50,
      "count_basis": "questions across public and private splits",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-seqqa-orf-seq-aaseq-creator-mcq"
      ],
      "evidence_ids": [
        "lab-bench-seqqa-orf-seq-aaseq-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Translation of a DNA open reading frame.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "rna-processing-translation-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-seqqa-orf-seq-numlen",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench SeqQA — ORF count above length",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "complete",
      "confidence": "high",
      "count": 50,
      "count_basis": "questions across public and private splits",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-seqqa-orf-seq-numlen-creator-mcq"
      ],
      "evidence_ids": [
        "lab-bench-seqqa-orf-seq-numlen-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Open-reading-frame interpretation.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "rna-processing-translation-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-seqqa-orf-transeff",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench SeqQA — Translation efficiency",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "complete",
      "confidence": "high",
      "count": 50,
      "count_basis": "questions across public and private splits",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-seqqa-orf-transeff-creator-mcq"
      ],
      "evidence_ids": [
        "lab-bench-seqqa-orf-transeff-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Translation-efficiency reasoning.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "rna-processing-translation-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "bioinstruction",
      "benchmark_kind": "suite",
      "benchmark_name": "Biology-Instructions",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 1,
      "count_basis": "Formal evaluation tracks (RNA-Protein Interaction Prediction).",
      "count_ref": null,
      "count_unit": "tracks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "bioinstruction-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Track count; not an example count.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "rna-protein-interaction-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "bioinstruction-rpi",
      "benchmark_kind": "track",
      "benchmark_name": "Biology-Instructions RNA-Protein Interaction Prediction",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 20824,
      "count_basis": "distinct examples across the published train, validation, and test splits",
      "count_ref": "/task_counts/total",
      "count_unit": "examples",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-rpi-closed-baselines",
        "bioinstruction-rpi-creator-systems",
        "bioinstruction-rpi-open-baselines"
      ],
      "evidence_ids": [
        "bioinstruction-rpi-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "RNA-protein interaction prediction.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "rna-protein-interaction-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "beacon-rna",
      "benchmark_kind": "suite",
      "benchmark_name": "BEACON",
      "benchmark_version": "neurips-2024",
      "classification_status": "complete",
      "confidence": "high",
      "count": 1,
      "count_basis": "Formal RNA benchmark tasks.",
      "count_ref": null,
      "count_unit": "tasks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "beacon-paper"
      ],
      "evaluation_run_ids": [
        "beacon-creator-full"
      ],
      "evidence_ids": [
        "beacon-paper-definition-evidence"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "Condition-specific RNA degradation prediction.",
      "reporting_status": "reported",
      "root_family_id": "beacon-rna",
      "task_type_id": "rna-stability-degradation-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "beacon-rna",
      "benchmark_kind": "suite",
      "benchmark_name": "BEACON",
      "benchmark_version": "neurips-2024",
      "classification_status": "complete",
      "confidence": "high",
      "count": 4,
      "count_basis": "Formal RNA benchmark tasks.",
      "count_ref": null,
      "count_unit": "tasks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "beacon-paper"
      ],
      "evaluation_run_ids": [
        "beacon-creator-full"
      ],
      "evidence_ids": [
        "beacon-paper-definition-evidence"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "Secondary structure, structural score, distance-map, and tertiary-structure tasks.",
      "reporting_status": "reported",
      "root_family_id": "beacon-rna",
      "task_type_id": "rna-structure-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": "2026-07-22",
      "audit_status": "audited",
      "benchmark_id": "atom3d",
      "benchmark_kind": "suite",
      "benchmark_name": "ATOM3D",
      "benchmark_version": "v0.2.6",
      "classification_status": "complete",
      "confidence": "high",
      "count": 1,
      "count_basis": "Curated 3D benchmark datasets.",
      "count_ref": null,
      "count_unit": "tasks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "atom3d-paper"
      ],
      "evaluation_run_ids": [
        "atom3d-creator-full"
      ],
      "evidence_ids": [
        "atom3d-paper-definition-evidence"
      ],
      "mapping_method": "official-track",
      "notes": "RSR ranks candidate RNA structures and is therefore not relabeled as de novo RNA folding.",
      "reporting_status": "reported",
      "root_family_id": "atom3d",
      "task_type_id": "rna-structure-quality-assessment"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "beacon-rna",
      "benchmark_kind": "suite",
      "benchmark_name": "BEACON",
      "benchmark_version": "neurips-2024",
      "classification_status": "complete",
      "confidence": "high",
      "count": 1,
      "count_basis": "Formal RNA benchmark tasks.",
      "count_ref": null,
      "count_unit": "tasks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "beacon-paper"
      ],
      "evaluation_run_ids": [
        "beacon-creator-full"
      ],
      "evidence_ids": [
        "beacon-paper-definition-evidence"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "Programmable RNA-switch activity prediction.",
      "reporting_status": "reported",
      "root_family_id": "beacon-rna",
      "task_type_id": "rna-switch-activity-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "lab-bench",
      "benchmark_kind": "suite",
      "benchmark_name": "LAB-Bench",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "partial",
      "confidence": "high",
      "count": 650,
      "count_basis": "Questions across the ten formal DbQA child tasks.",
      "count_ref": null,
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "lab-bench-evidence-repository"
      ],
      "mapping_method": "official-track",
      "notes": "Count is specific to DbQA and is not added to other task claims.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "scientific-database-retrieval"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-dbqa",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench DbQA",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "complete",
      "confidence": "high",
      "count": 650,
      "count_basis": "questions across formal child tasks",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "lab-bench-dbqa-evidence-repository"
      ],
      "mapping_method": "official-track",
      "notes": "Complete formal DbQA category.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "scientific-database-retrieval"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-dbqa-dga",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench DbQA — Disease gene associations",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "complete",
      "confidence": "high",
      "count": 50,
      "count_basis": "questions across public and private splits",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-dbqa-dga-creator-mcq"
      ],
      "evidence_ids": [
        "lab-bench-dbqa-dga-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "DisGeNET and OMIM retrieval.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "scientific-database-retrieval"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-dbqa-gene-location",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench DbQA — Gene location",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "complete",
      "confidence": "high",
      "count": 50,
      "count_basis": "questions across public and private splits",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-dbqa-gene-location-creator-mcq"
      ],
      "evidence_ids": [
        "lab-bench-dbqa-gene-location-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Ensembl gene-location retrieval.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "scientific-database-retrieval"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-dbqa-mirna-targets",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench DbQA — miRNA targets",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "complete",
      "confidence": "high",
      "count": 50,
      "count_basis": "questions across public and private splits",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-dbqa-mirna-targets-creator-mcq"
      ],
      "evidence_ids": [
        "lab-bench-dbqa-mirna-targets-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "miRDB target retrieval.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "scientific-database-retrieval"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-dbqa-mouse-tumor-gene-sets",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench DbQA — Mouse tumor gene sets",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "complete",
      "confidence": "high",
      "count": 100,
      "count_basis": "questions across public and private splits",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-dbqa-mouse-tumor-gene-sets-creator-mcq"
      ],
      "evidence_ids": [
        "lab-bench-dbqa-mouse-tumor-gene-sets-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Gene-set membership retrieval.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "scientific-database-retrieval"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-dbqa-oncogenic-signatures",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench DbQA — Oncogenic signatures",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "complete",
      "confidence": "high",
      "count": 50,
      "count_basis": "questions across public and private splits",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-dbqa-oncogenic-signatures-creator-mcq"
      ],
      "evidence_ids": [
        "lab-bench-dbqa-oncogenic-signatures-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "MSigDB signature retrieval.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "scientific-database-retrieval"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-dbqa-tfbs-gtrd",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench DbQA — GTRD transcription-factor binding sites",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "complete",
      "confidence": "high",
      "count": 50,
      "count_basis": "questions across public and private splits",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-dbqa-tfbs-gtrd-creator-mcq"
      ],
      "evidence_ids": [
        "lab-bench-dbqa-tfbs-gtrd-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Database retrieval is the evaluated operation.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "scientific-database-retrieval"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-dbqa-variant-from-sequence",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench DbQA — Protein variant from sequence",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "complete",
      "confidence": "high",
      "count": 100,
      "count_basis": "questions across public and private splits",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-dbqa-variant-from-sequence-creator-mcq"
      ],
      "evidence_ids": [
        "lab-bench-dbqa-variant-from-sequence-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "ClinVar retrieval.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "scientific-database-retrieval"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-dbqa-variant-multi-sequence",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench DbQA — Protein variant with multiple sequences",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "complete",
      "confidence": "high",
      "count": 100,
      "count_basis": "questions across public and private splits",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-dbqa-variant-multi-sequence-creator-mcq"
      ],
      "evidence_ids": [
        "lab-bench-dbqa-variant-multi-sequence-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "ClinVar retrieval.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "scientific-database-retrieval"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-dbqa-vax-response",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench DbQA — Vaccine response gene sets",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "complete",
      "confidence": "high",
      "count": 50,
      "count_basis": "questions across public and private splits",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-dbqa-vax-response-creator-mcq"
      ],
      "evidence_ids": [
        "lab-bench-dbqa-vax-response-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Vaccine-response gene-set retrieval.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "scientific-database-retrieval"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-dbqa-viral-ppi",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench DbQA — Viral protein–protein interactions",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "complete",
      "confidence": "high",
      "count": 50,
      "count_basis": "questions across public and private splits",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-dbqa-viral-ppi-creator-mcq"
      ],
      "evidence_ids": [
        "lab-bench-dbqa-viral-ppi-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Database retrieval is part of the evaluated task.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "scientific-database-retrieval"
    },
    {
      "aggregate_eligible": false,
      "as_of": "2026-01-11",
      "audit_status": "audited",
      "benchmark_id": "anthropic-scientific-figure-interpretation",
      "benchmark_kind": "track",
      "benchmark_name": "Anthropic Scientific Figure Interpretation Eval",
      "benchmark_version": "reported-2026-01-11",
      "classification_status": "complete",
      "confidence": "high",
      "count": null,
      "count_basis": "private scientific figure interpretation tasks",
      "count_ref": null,
      "count_unit": "tasks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "anthropic-healthcare-life-sciences"
      ],
      "evaluation_run_ids": [
        "anthropic-scientific-figure-delta"
      ],
      "evidence_ids": [
        "anthropic-scientific-figure-evidence"
      ],
      "mapping_method": "official-track",
      "notes": "No task-level examples or count are public.",
      "reporting_status": "not_reported",
      "root_family_id": "anthropic-key-life-sciences-evals",
      "task_type_id": "scientific-evidence-interpretation"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "blade-mcq",
      "benchmark_kind": "track",
      "benchmark_name": "BLADE Decision-Discrimination MCQ",
      "benchmark_version": "arXiv v3",
      "classification_status": "complete",
      "confidence": "high",
      "count": 188,
      "count_basis": "individual multiple-choice decision-discrimination questions",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "blade-paper"
      ],
      "evaluation_run_ids": [
        "blade-creator-decision-mcq"
      ],
      "evidence_ids": [
        "blade-mcq-evidence-counts"
      ],
      "mapping_method": "official-track",
      "notes": "Questions test defensibility of scientific analysis decisions.",
      "reporting_status": "reported",
      "root_family_id": "blade",
      "task_type_id": "scientific-evidence-interpretation"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "lab-bench",
      "benchmark_kind": "suite",
      "benchmark_name": "LAB-Bench",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "partial",
      "confidence": "high",
      "count": null,
      "count_basis": "FigQA, LitQA2, SuppQA, and TableQA questions.",
      "count_ref": null,
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "lab-bench-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "The root record does not publish this cross-track subtotal.",
      "reporting_status": "not_reported",
      "root_family_id": "lab-bench",
      "task_type_id": "scientific-evidence-interpretation"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-figqa",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench FigQA",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "complete",
      "confidence": "high",
      "count": 226,
      "count_basis": "questions across public and private splits",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "anthropic-sonnet-4-5-system-card",
        "anthropic-sonnet-4-6-system-card",
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-figqa-anthropic-sonnet45-system-card",
        "lab-bench-figqa-creator-mcq",
        "lab-bench-figqa-creator-open-response",
        "lab-bench-figqa-crop-tool",
        "lab-bench-figqa-no-tools"
      ],
      "evidence_ids": [
        "lab-bench-figqa-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Scientific figure interpretation.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "scientific-evidence-interpretation"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-litqa2",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench LitQA2",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "complete",
      "confidence": "high",
      "count": 248,
      "count_basis": "questions across public and private splits",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-litqa2-creator-mcq"
      ],
      "evidence_ids": [
        "lab-bench-litqa2-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Full-paper evidence retrieval and interpretation.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "scientific-evidence-interpretation"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-suppqa",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench SuppQA",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "complete",
      "confidence": "high",
      "count": 102,
      "count_basis": "questions across public and private splits",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-suppqa-creator-mcq"
      ],
      "evidence_ids": [
        "lab-bench-suppqa-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Supplementary text and table interpretation.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "scientific-evidence-interpretation"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-tableqa",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench TableQA",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "complete",
      "confidence": "high",
      "count": 305,
      "count_basis": "questions across public and private splits",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-tableqa-creator-mcq"
      ],
      "evidence_ids": [
        "lab-bench-tableqa-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Scientific table interpretation.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "scientific-evidence-interpretation"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "scigym",
      "benchmark_kind": "suite",
      "benchmark_name": "SCIGYM",
      "benchmark_version": "2025 release",
      "classification_status": "complete",
      "confidence": "high",
      "count": 350,
      "count_basis": "distinct curated BioModels systems released as SBML benchmark instances",
      "count_ref": "/task_counts/total",
      "count_unit": "systems",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "scigym-evidence-release-counts",
        "scigym-evidence-taxonomy"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "The same systems support iterative in-silico perturbation experiments; overlapping claims are not summed.",
      "reporting_status": "reported",
      "root_family_id": "scigym",
      "task_type_id": "simulation-based-experiment"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "scigym-large",
      "benchmark_kind": "track",
      "benchmark_name": "SCIGYM Large",
      "benchmark_version": "2025 release",
      "classification_status": "complete",
      "confidence": "high",
      "count": 213,
      "count_basis": "unique SBML systems in the official large Parquet split, containing the remaining systems with up to 400 reactions",
      "count_ref": "/task_counts/total",
      "count_unit": "systems",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "scigym-large-evidence-count"
      ],
      "mapping_method": "official-track",
      "notes": "Same systems; overlapping task claim.",
      "reporting_status": "reported",
      "root_family_id": "scigym",
      "task_type_id": "simulation-based-experiment"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "scigym-small",
      "benchmark_kind": "track",
      "benchmark_name": "SCIGYM Small",
      "benchmark_version": "2025 release",
      "classification_status": "complete",
      "confidence": "high",
      "count": 137,
      "count_basis": "unique SBML systems with fewer than 10 reactions in the official small Parquet split",
      "count_ref": "/task_counts/total",
      "count_unit": "systems",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "scigym-paper"
      ],
      "evaluation_run_ids": [
        "scigym-small-creator-paper",
        "scigym-small-zero-shot"
      ],
      "evidence_ids": [
        "scigym-small-evidence-count"
      ],
      "mapping_method": "official-track",
      "notes": "Same systems; overlapping task claim.",
      "reporting_status": "reported",
      "root_family_id": "scigym",
      "task_type_id": "simulation-based-experiment"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "scib",
      "benchmark_kind": "suite",
      "benchmark_name": "scIB",
      "benchmark_version": "paper-2021",
      "classification_status": "complete",
      "confidence": "high",
      "count": 13,
      "count_basis": "Atlas-level integration tasks.",
      "count_ref": "/task_counts/total",
      "count_unit": "tasks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "scib-paper"
      ],
      "evaluation_run_ids": [
        "scib-creator-full"
      ],
      "evidence_ids": [
        "scib-paper-definition-evidence"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "Two simulation, five scRNA-seq, and six scATAC-seq integration tasks.",
      "reporting_status": "reported",
      "root_family_id": "scib",
      "task_type_id": "single-cell-data-integration"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "bioinstruction",
      "benchmark_kind": "suite",
      "benchmark_name": "Biology-Instructions",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 1,
      "count_basis": "Formal evaluation tracks (siRNA Efficiency Prediction).",
      "count_ref": null,
      "count_unit": "tracks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "bioinstruction-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Track count; not an example count.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "sirna-efficacy-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "bioinstruction-sirna",
      "benchmark_kind": "track",
      "benchmark_name": "Biology-Instructions siRNA Efficiency Prediction",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 66987,
      "count_basis": "distinct examples across the published train, validation, and test splits",
      "count_ref": "/task_counts/total",
      "count_unit": "examples",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-sirna-closed-baselines",
        "bioinstruction-sirna-creator-systems",
        "bioinstruction-sirna-open-baselines"
      ],
      "evidence_ids": [
        "bioinstruction-sirna-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "siRNA efficiency prediction.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "sirna-efficacy-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "guacamol",
      "benchmark_kind": "suite",
      "benchmark_name": "GuacaMol",
      "benchmark_version": "suite-v2",
      "classification_status": "complete",
      "confidence": "high",
      "count": 25,
      "count_basis": "Formal benchmark problems.",
      "count_ref": "/task_counts/total",
      "count_unit": "problems",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "guacamol-paper"
      ],
      "evaluation_run_ids": [
        "guacamol-creator-full"
      ],
      "evidence_ids": [
        "guacamol-paper-definition-evidence",
        "guacamol-repository-evidence"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "Includes both distribution-learning assessment and goal-directed molecular optimization problems.",
      "reporting_status": "reported",
      "root_family_id": "guacamol",
      "task_type_id": "small-molecule-generation"
    },
    {
      "aggregate_eligible": true,
      "as_of": "2026-07-22",
      "audit_status": "audited",
      "benchmark_id": "atom3d",
      "benchmark_kind": "suite",
      "benchmark_name": "ATOM3D",
      "benchmark_version": "v0.2.6",
      "classification_status": "complete",
      "confidence": "high",
      "count": 1,
      "count_basis": "Curated 3D benchmark datasets.",
      "count_ref": null,
      "count_unit": "tasks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "atom3d-paper"
      ],
      "evaluation_run_ids": [
        "atom3d-creator-full"
      ],
      "evidence_ids": [
        "atom3d-paper-definition-evidence"
      ],
      "mapping_method": "official-track",
      "notes": "SMP, small-molecule property prediction from molecular structure.",
      "reporting_status": "reported",
      "root_family_id": "atom3d",
      "task_type_id": "small-molecule-property-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "moleculenet",
      "benchmark_kind": "suite",
      "benchmark_name": "MoleculeNet",
      "benchmark_version": "original-2017",
      "classification_status": "partial",
      "confidence": "high",
      "count": 17,
      "count_basis": "Original paper dataset collections.",
      "count_ref": "/task_counts/total",
      "count_unit": "other",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "enhancing-molecular-property-prediction-with-auxiliary",
        "moleculenet-paper"
      ],
      "evaluation_run_ids": [
        "moleculenet-creator-full"
      ],
      "evidence_ids": [
        "moleculenet-paper-definition-evidence"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "Umbrella mapping for all 17 original dataset collections, which span quantum, physical, biophysical, and physiological properties.",
      "reporting_status": "reported",
      "root_family_id": "moleculenet",
      "task_type_id": "small-molecule-property-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": "2026-06-10",
      "audit_status": "audited-with-caveats",
      "benchmark_id": "spatialbench",
      "benchmark_kind": "agentic-eval",
      "benchmark_name": "SpatialBench",
      "benchmark_version": "repo-159-5042c4f",
      "classification_status": "partial",
      "confidence": "high",
      "count": 36,
      "count_basis": "official category_results.json n_evals",
      "count_ref": "/task_counts/subsets/6/count",
      "count_unit": "problems",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "spatialbench-preprint",
        "spatialbench-repository-release",
        "system-card-claude-opus-5"
      ],
      "evaluation_run_ids": [
        "spatialbench-paper-v2-base",
        "spatialbench-paper-v2-claude-code",
        "spatialbench-paper-v2-latch",
        "spatialbench-repo-159-claude-code",
        "spatialbench-repo-159-mini-swe-agent",
        "spatialbench-repo-159-openai-codex",
        "spatialbench-repo-159-pi"
      ],
      "evidence_ids": [
        "spatialbench-evidence-current-counts"
      ],
      "mapping_method": "official-taxonomy",
      "notes": "Official Spatial Analysis category.",
      "reporting_status": "reported",
      "root_family_id": "spatialbench",
      "task_type_id": "spatial-omics-analysis"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited-with-caveats",
      "benchmark_id": "bioinstruction",
      "benchmark_kind": "suite",
      "benchmark_name": "Biology-Instructions",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 2,
      "count_basis": "Formal evaluation tracks (Human and mouse Transcription Binding Sites Detection).",
      "count_ref": null,
      "count_unit": "tracks",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [],
      "evaluation_run_ids": [],
      "evidence_ids": [
        "bioinstruction-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Track count; not an example count.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "transcription-factor-binding-site-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "bioinstruction-tb-human",
      "benchmark_kind": "track",
      "benchmark_name": "Biology-Instructions Human Transcription Binding Sites Detection",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 138344,
      "count_basis": "distinct examples across the published train, validation, and test splits",
      "count_ref": "/task_counts/total",
      "count_unit": "examples",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-tb-human-closed-baselines",
        "bioinstruction-tb-human-creator-systems",
        "bioinstruction-tb-human-open-baselines"
      ],
      "evidence_ids": [
        "bioinstruction-tb-human-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Human transcription-factor binding-site detection.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "transcription-factor-binding-site-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "bioinstruction-tb-mouse",
      "benchmark_kind": "track",
      "benchmark_name": "Biology-Instructions Mouse Transcription Binding Sites Detection",
      "benchmark_version": "emnlp-2025",
      "classification_status": "complete",
      "confidence": "high",
      "count": 100028,
      "count_basis": "distinct examples across the published train, validation, and test splits",
      "count_ref": "/task_counts/total",
      "count_unit": "examples",
      "coverage": "explicitly-in-scope",
      "evaluating_work_ids": [
        "bioinstructions-paper"
      ],
      "evaluation_run_ids": [
        "bioinstruction-tb-mouse-closed-baselines",
        "bioinstruction-tb-mouse-creator-systems",
        "bioinstruction-tb-mouse-open-baselines"
      ],
      "evidence_ids": [
        "bioinstruction-tb-mouse-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "Mouse transcription-factor binding-site detection.",
      "reporting_status": "reported",
      "root_family_id": "bioinstruction",
      "task_type_id": "transcription-factor-binding-site-prediction"
    },
    {
      "aggregate_eligible": true,
      "as_of": null,
      "audit_status": "audited",
      "benchmark_id": "lab-bench-dbqa-tfbs-gtrd",
      "benchmark_kind": "track",
      "benchmark_name": "LAB-Bench DbQA — GTRD transcription-factor binding sites",
      "benchmark_version": "repository-998a8e0",
      "classification_status": "complete",
      "confidence": "high",
      "count": 50,
      "count_basis": "questions across public and private splits",
      "count_ref": "/task_counts/total",
      "count_unit": "questions",
      "coverage": "observed",
      "evaluating_work_ids": [
        "lab-bench-paper"
      ],
      "evaluation_run_ids": [
        "lab-bench-dbqa-tfbs-gtrd-creator-mcq"
      ],
      "evidence_ids": [
        "lab-bench-dbqa-tfbs-gtrd-evidence-paper"
      ],
      "mapping_method": "official-track",
      "notes": "The artifact concerns TFBS annotations; it does not evaluate a de novo predictor.",
      "reporting_status": "reported",
      "root_family_id": "lab-bench",
      "task_type_id": "transcription-factor-binding-site-prediction"
    }
  ],
  "scientific_tasks": [
    {
      "aliases": [
        "protein structure"
      ],
      "coverage_family_count": 1,
      "coverage_track_count": 0,
      "definition": "Predict or assess the three-dimensional structure of proteins or protein assemblies.",
      "deprecated_aliases": [],
      "id": "protein-structure",
      "label": "Protein structure prediction",
      "label_zh": "蛋白质结构预测",
      "object_ids": [
        "protein"
      ],
      "official_work_count": 1,
      "parent_id": null,
      "task_family_id": "structure-prediction"
    },
    {
      "aliases": [
        "protein folding",
        "folding",
        "monomer prediction"
      ],
      "coverage_family_count": 1,
      "coverage_track_count": 1,
      "definition": "Predict the three-dimensional structure of a single protein chain or domain.",
      "deprecated_aliases": [],
      "id": "protein-monomer-structure-prediction",
      "label": "Protein monomer structure prediction",
      "label_zh": "蛋白质单体折叠与结构预测",
      "object_ids": [
        "protein"
      ],
      "official_work_count": 1,
      "parent_id": "protein-structure",
      "task_family_id": "structure-prediction"
    },
    {
      "aliases": [
        "complex prediction",
        "multimer prediction",
        "protein docking"
      ],
      "coverage_family_count": 2,
      "coverage_track_count": 1,
      "definition": "Predict the assembled structure or stoichiometry of a protein complex.",
      "deprecated_aliases": [],
      "id": "protein-complex-structure-prediction",
      "label": "Protein complex structure prediction",
      "label_zh": "蛋白质复合物结构预测",
      "object_ids": [
        "protein"
      ],
      "official_work_count": 2,
      "parent_id": "protein-structure",
      "task_family_id": "structure-prediction"
    },
    {
      "aliases": [
        "model accuracy estimation",
        "structure quality assessment"
      ],
      "coverage_family_count": 3,
      "coverage_track_count": 1,
      "definition": "Estimate the accuracy or quality of a predicted protein structure model.",
      "deprecated_aliases": [],
      "id": "protein-model-quality-assessment",
      "label": "Protein model quality assessment",
      "label_zh": "蛋白质模型质量评估",
      "object_ids": [
        "protein"
      ],
      "official_work_count": 3,
      "parent_id": "protein-structure",
      "task_family_id": "structure-prediction"
    },
    {
      "aliases": [
        "secondary structure prediction",
        "SSP"
      ],
      "coil": null,
      "coverage_family_count": 1,
      "coverage_track_count": 0,
      "definition": "Predict per-residue helix",
      "deprecated_aliases": [],
      "id": "protein-secondary-structure-prediction",
      "label": "Protein secondary-structure prediction",
      "label_zh": "蛋白质二级结构预测",
      "object_ids": [
        "protein"
      ],
      "official_work_count": 1,
      "or other secondary-structure labels from a protein sequence.": null,
      "parent_id": "protein-structure",
      "strand": null,
      "task_family_id": "structure-prediction"
    },
    {
      "aliases": [
        "contact prediction",
        "residue contact prediction"
      ],
      "coverage_family_count": 1,
      "coverage_track_count": 0,
      "definition": "Predict which residue pairs in a protein are spatially in contact.",
      "deprecated_aliases": [],
      "id": "protein-contact-map-prediction",
      "label": "Protein contact-map prediction",
      "label_zh": "蛋白质接触图预测",
      "object_ids": [
        "protein"
      ],
      "official_work_count": 1,
      "parent_id": "protein-structure",
      "task_family_id": "structure-prediction"
    },
    {
      "aliases": [
        "protein engineering"
      ],
      "coverage_family_count": 1,
      "coverage_track_count": 0,
      "definition": "Generate or optimize proteins under structural",
      "deprecated_aliases": [],
      "functional": null,
      "id": "protein-design",
      "label": "Protein design",
      "label_zh": "蛋白质设计",
      "object_ids": [
        "protein"
      ],
      "official_work_count": 1,
      "or fitness constraints.": null,
      "parent_id": null,
      "task_family_id": "design-generation"
    },
    {
      "aliases": [
        "sequence generation",
        "inverse folding"
      ],
      "coverage_family_count": 3,
      "coverage_track_count": 0,
      "definition": "Generate an amino-acid sequence satisfying specified structural or functional constraints.",
      "deprecated_aliases": [],
      "id": "protein-sequence-design",
      "label": "Protein sequence design",
      "label_zh": "蛋白质序列设计",
      "object_ids": [
        "protein"
      ],
      "official_work_count": 1,
      "parent_id": "protein-design",
      "task_family_id": "design-generation"
    },
    {
      "aliases": [
        "fitness optimization",
        "directed evolution design"
      ],
      "coverage_family_count": 0,
      "coverage_track_count": 0,
      "definition": "Select or optimize protein variants toward improved measured fitness.",
      "deprecated_aliases": [],
      "id": "protein-fitness-optimization",
      "label": "Protein fitness optimization",
      "label_zh": "蛋白质适应度优化",
      "object_ids": [
        "protein"
      ],
      "official_work_count": 0,
      "parent_id": "protein-design",
      "task_family_id": "design-generation"
    },
    {
      "aliases": [
        "protein property prediction"
      ],
      "coverage_family_count": 0,
      "coverage_track_count": 0,
      "definition": "Predict protein function",
      "deprecated_aliases": [],
      "fitness": null,
      "id": "protein-property-function",
      "label": "Protein property and function prediction",
      "label_zh": "蛋白质性质与功能预测",
      "object_ids": [
        "protein"
      ],
      "official_work_count": 0,
      "or clinically relevant effects.": null,
      "parent_id": null,
      "solubility": null,
      "stability": null,
      "task_family_id": "property-function-prediction"
    },
    {
      "aliases": [
        "enzyme function prediction",
        "EC prediction"
      ],
      "coverage_family_count": 1,
      "coverage_track_count": 1,
      "definition": "Assign molecular function",
      "deprecated_aliases": [],
      "enzyme class": null,
      "id": "protein-function-annotation",
      "label": "Protein function annotation",
      "label_zh": "蛋白质功能注释",
      "object_ids": [
        "protein"
      ],
      "official_work_count": 1,
      "or functional labels to a protein.": null,
      "parent_id": "protein-property-function",
      "task_family_id": "property-function-prediction"
    },
    {
      "aliases": [
        "remote homology",
        "fold recognition"
      ],
      "coverage_family_count": 1,
      "coverage_track_count": 0,
      "definition": "Detect structural or evolutionary relationships between distantly related protein sequences.",
      "deprecated_aliases": [],
      "id": "protein-remote-homology-detection",
      "label": "Protein remote-homology detection",
      "label_zh": "蛋白质远缘同源检测",
      "object_ids": [
        "protein"
      ],
      "official_work_count": 1,
      "parent_id": "protein-property-function",
      "task_family_id": "property-function-prediction"
    },
    {
      "aliases": [
        "residue identity",
        "amino-acid environment classification"
      ],
      "coverage_family_count": 1,
      "coverage_track_count": 0,
      "definition": "Predict the amino-acid identity associated with a local protein structural environment.",
      "deprecated_aliases": [],
      "id": "protein-residue-identity-prediction",
      "label": "Protein residue-identity prediction",
      "label_zh": "蛋白质残基身份预测",
      "object_ids": [
        "protein"
      ],
      "official_work_count": 1,
      "parent_id": "protein-property-function",
      "task_family_id": "property-function-prediction"
    },
    {
      "aliases": [
        "fluorescence prediction"
      ],
      "coverage_family_count": 2,
      "coverage_track_count": 1,
      "definition": "Predict a measured fluorescence phenotype from a protein sequence.",
      "deprecated_aliases": [],
      "id": "protein-fluorescence-prediction",
      "label": "Protein fluorescence prediction",
      "label_zh": "蛋白质荧光性质预测",
      "object_ids": [
        "protein"
      ],
      "official_work_count": 2,
      "parent_id": "protein-property-function",
      "task_family_id": "property-function-prediction"
    },
    {
      "aliases": [
        "variant effect prediction",
        "mutation effect"
      ],
      "coverage_family_count": 2,
      "coverage_track_count": 2,
      "definition": "Predict the functional or phenotypic effect of protein sequence variants.",
      "deprecated_aliases": [],
      "id": "protein-mutation-effect-prediction",
      "label": "Protein mutation-effect prediction",
      "label_zh": "蛋白质突变效应预测",
      "object_ids": [
        "protein"
      ],
      "official_work_count": 2,
      "parent_id": "protein-property-function",
      "task_family_id": "variant-perturbation"
    },
    {
      "aliases": [
        "fitness landscape prediction",
        "sequence-to-fitness"
      ],
      "coverage_family_count": 2,
      "coverage_track_count": 4,
      "definition": "Predict a quantitative or ranked protein fitness measurement from sequence.",
      "deprecated_aliases": [],
      "id": "protein-fitness-prediction",
      "label": "Protein fitness prediction",
      "label_zh": "蛋白质适应度预测",
      "object_ids": [
        "protein"
      ],
      "official_work_count": 3,
      "parent_id": "protein-property-function",
      "task_family_id": "property-function-prediction"
    },
    {
      "aliases": [
        "thermostability prediction",
        "melting temperature prediction"
      ],
      "coverage_family_count": 3,
      "coverage_track_count": 3,
      "definition": "Predict stability",
      "deprecated_aliases": [],
      "id": "protein-stability-prediction",
      "label": "Protein stability prediction",
      "label_zh": "蛋白质稳定性预测",
      "object_ids": [
        "protein"
      ],
      "official_work_count": 3,
      "or melting behavior of a protein.": null,
      "parent_id": "protein-property-function",
      "task_family_id": "property-function-prediction",
      "thermostability": null
    },
    {
      "aliases": [
        "solubility"
      ],
      "coverage_family_count": 1,
      "coverage_track_count": 1,
      "definition": "Predict whether or how strongly a protein is soluble under an assay condition.",
      "deprecated_aliases": [],
      "id": "protein-solubility-prediction",
      "label": "Protein solubility prediction",
      "label_zh": "蛋白质溶解性预测",
      "object_ids": [
        "protein"
      ],
      "official_work_count": 1,
      "parent_id": "protein-property-function",
      "task_family_id": "property-function-prediction"
    },
    {
      "aliases": [
        "variant pathogenicity",
        "clinical variant classification"
      ],
      "coverage_family_count": 2,
      "coverage_track_count": 4,
      "definition": "Classify or prioritize protein variants for clinical relevance or pathogenicity.",
      "deprecated_aliases": [],
      "id": "protein-clinical-variant-interpretation",
      "label": "Protein clinical variant interpretation",
      "label_zh": "蛋白质临床变异解读",
      "object_ids": [
        "protein"
      ],
      "official_work_count": 2,
      "parent_id": "protein-property-function",
      "task_family_id": "variant-perturbation"
    },
    {
      "aliases": [
        "molecular binding"
      ],
      "binding geometry": null,
      "coverage_family_count": 1,
      "coverage_track_count": 0,
      "definition": "Predict or analyze interactions",
      "deprecated_aliases": [],
      "id": "molecular-interaction-analysis",
      "label": "Molecular interaction and binding",
      "label_zh": "分子相互作用与结合",
      "object_ids": [
        "protein",
        "rna",
        "small-molecule"
      ],
      "official_work_count": 1,
      "or affinity between molecules.": null,
      "parent_id": null,
      "specificity": null,
      "task_family_id": "molecular-interaction"
    },
    {
      "aliases": [
        "PPI",
        "protein binding protein"
      ],
      "coverage_family_count": 1,
      "coverage_track_count": 1,
      "definition": "Predict whether proteins interact or identify an interaction partner or interface.",
      "deprecated_aliases": [],
      "id": "protein-protein-interaction-prediction",
      "label": "Protein-protein interaction prediction",
      "label_zh": "蛋白质-蛋白质相互作用预测",
      "object_ids": [
        "protein"
      ],
      "official_work_count": 1,
      "parent_id": "molecular-interaction-analysis",
      "task_family_id": "molecular-interaction"
    },
    {
      "aliases": [
        "protein interface prediction",
        "interface contact prediction"
      ],
      "coverage_family_count": 1,
      "coverage_track_count": 0,
      "definition": "Predict which residues from two proteins will contact one another when the proteins bind.",
      "deprecated_aliases": [],
      "id": "protein-protein-interface-prediction",
      "label": "Protein-protein interface prediction",
      "label_zh": "蛋白质-蛋白质界面预测",
      "object_ids": [
        "protein"
      ],
      "official_work_count": 1,
      "parent_id": "molecular-interaction-analysis",
      "task_family_id": "molecular-interaction"
    },
    {
      "aliases": [
        "mutation stability prediction",
        "complex stability change"
      ],
      "coverage_family_count": 1,
      "coverage_track_count": 0,
      "definition": "Predict whether a mutation increases or decreases the stability of a protein complex or interaction.",
      "deprecated_aliases": [],
      "id": "protein-complex-mutation-stability-prediction",
      "label": "Protein-complex mutation stability prediction",
      "label_zh": "蛋白质复合物突变稳定性预测",
      "object_ids": [
        "protein"
      ],
      "official_work_count": 1,
      "parent_id": "molecular-interaction-analysis",
      "task_family_id": "variant-perturbation"
    },
    {
      "aliases": [
        "PPI affinity"
      ],
      "coverage_family_count": 0,
      "coverage_track_count": 0,
      "definition": "Predict or rank the strength of binding between protein partners.",
      "deprecated_aliases": [],
      "id": "protein-protein-binding-affinity",
      "label": "Protein-protein binding affinity",
      "label_zh": "蛋白质-蛋白质结合亲和力",
      "object_ids": [
        "protein"
      ],
      "official_work_count": 0,
      "parent_id": "molecular-interaction-analysis",
      "task_family_id": "molecular-interaction"
    },
    {
      "aliases": [
        "antibody binding",
        "antigen binding",
        "neutralization"
      ],
      "coverage_family_count": 2,
      "coverage_track_count": 2,
      "definition": "Predict antibody-antigen recognition",
      "deprecated_aliases": [],
      "id": "antibody-antigen-interaction",
      "label": "Antibody-antigen interaction",
      "label_zh": "抗体-抗原相互作用",
      "neutralization": null,
      "object_ids": [
        "protein"
      ],
      "official_work_count": 1,
      "or affinity.": null,
      "parent_id": "molecular-interaction-analysis",
      "specificity": null,
      "structure": null,
      "task_family_id": "molecular-interaction"
    },
    {
      "aliases": [
        "protein ligand binding",
        "protein-small-molecule binding",
        "ligand binding"
      ],
      "coverage_family_count": 1,
      "coverage_track_count": 0,
      "definition": "Predict or reason about protein-ligand binding when the official source does not distinguish pose from affinity.",
      "deprecated_aliases": [],
      "id": "protein-ligand-binding-prediction",
      "label": "Protein-ligand binding prediction",
      "label_zh": "蛋白质-配体结合预测",
      "object_ids": [
        "protein",
        "small-molecule"
      ],
      "official_work_count": 1,
      "parent_id": "molecular-interaction-analysis",
      "task_family_id": "molecular-interaction"
    },
    {
      "aliases": [
        "ligand efficacy prediction",
        "ligand activation prediction"
      ],
      "coverage_family_count": 1,
      "coverage_track_count": 0,
      "definition": "Given a ligand and target structures, predict whether the ligand activates or inhibits the protein's function rather than whether it binds.",
      "deprecated_aliases": [],
      "id": "protein-ligand-efficacy-prediction",
      "label": "Protein-ligand functional efficacy prediction",
      "label_zh": "蛋白质-配体功能效应预测",
      "object_ids": [
        "protein",
        "small-molecule"
      ],
      "official_work_count": 1,
      "parent_id": "molecular-interaction-analysis",
      "task_family_id": "property-function-prediction"
    },
    {
      "aliases": [
        "molecular docking",
        "ligand docking",
        "pose prediction"
      ],
      "coverage_family_count": 2,
      "coverage_track_count": 1,
      "definition": "Predict the bound pose or geometry of a small molecule in a protein complex.",
      "deprecated_aliases": [],
      "id": "protein-ligand-pose-prediction",
      "label": "Protein-ligand pose prediction",
      "label_zh": "蛋白质-配体结合构象预测",
      "object_ids": [
        "protein",
        "small-molecule"
      ],
      "official_work_count": 2,
      "parent_id": "molecular-interaction-analysis",
      "task_family_id": "molecular-interaction"
    },
    {
      "aliases": [
        "ligand affinity",
        "binding affinity"
      ],
      "coverage_family_count": 3,
      "coverage_track_count": 1,
      "definition": "Predict or rank the strength of protein-small-molecule binding.",
      "deprecated_aliases": [],
      "id": "protein-ligand-binding-affinity",
      "label": "Protein-ligand binding affinity",
      "label_zh": "蛋白质-配体结合亲和力",
      "object_ids": [
        "protein",
        "small-molecule"
      ],
      "official_work_count": 4,
      "parent_id": "molecular-interaction-analysis",
      "task_family_id": "molecular-interaction"
    },
    {
      "aliases": [
        "RPI",
        "RNA protein binding"
      ],
      "coverage_family_count": 1,
      "coverage_track_count": 1,
      "definition": "Predict whether an RNA and protein interact or bind.",
      "deprecated_aliases": [],
      "id": "rna-protein-interaction-prediction",
      "label": "RNA-protein interaction prediction",
      "label_zh": "RNA-蛋白质相互作用预测",
      "object_ids": [
        "rna",
        "protein"
      ],
      "official_work_count": 1,
      "parent_id": "molecular-interaction-analysis",
      "task_family_id": "molecular-interaction"
    },
    {
      "aliases": [
        "regulatory genomics"
      ],
      "chromatin signals": null,
      "coverage_family_count": 0,
      "coverage_track_count": 0,
      "definition": "Predict DNA regulatory elements",
      "deprecated_aliases": [],
      "id": "dna-regulation-perturbation",
      "label": "DNA regulation and perturbation",
      "label_zh": "DNA调控与扰动",
      "object_ids": [
        "dna"
      ],
      "official_work_count": 0,
      "or editing outcomes.": null,
      "parent_id": null,
      "task_family_id": "sequence-regulation",
      "variants": null
    },
    {
      "aliases": [
        "core promoter prediction"
      ],
      "coverage_family_count": 2,
      "coverage_track_count": 2,
      "definition": "Identify promoter regions or promoter activity from DNA sequence.",
      "deprecated_aliases": [],
      "id": "promoter-detection",
      "label": "Promoter detection",
      "label_zh": "启动子识别",
      "object_ids": [
        "dna"
      ],
      "official_work_count": 2,
      "parent_id": "dna-regulation-perturbation",
      "task_family_id": "sequence-regulation"
    },
    {
      "aliases": [
        "enhancer prediction"
      ],
      "coverage_family_count": 2,
      "coverage_track_count": 1,
      "definition": "Predict enhancer activity or regulatory potential from DNA sequence.",
      "deprecated_aliases": [],
      "id": "enhancer-activity-prediction",
      "label": "Enhancer activity prediction",
      "label_zh": "增强子活性预测",
      "object_ids": [
        "dna"
      ],
      "official_work_count": 2,
      "parent_id": "dna-regulation-perturbation",
      "task_family_id": "sequence-regulation"
    },
    {
      "aliases": [
        "TFBS prediction"
      ],
      "coverage_family_count": 2,
      "coverage_track_count": 3,
      "definition": "Predict transcription-factor binding sites or occupancy from sequence or genomic context.",
      "deprecated_aliases": [],
      "id": "transcription-factor-binding-site-prediction",
      "label": "Transcription-factor binding-site prediction",
      "label_zh": "转录因子结合位点预测",
      "object_ids": [
        "dna",
        "protein"
      ],
      "official_work_count": 2,
      "parent_id": "dna-regulation-perturbation",
      "task_family_id": "molecular-interaction"
    },
    {
      "aliases": [
        "EPI prediction"
      ],
      "coverage_family_count": 1,
      "coverage_track_count": 1,
      "definition": "Predict physical or functional enhancer-promoter interactions.",
      "deprecated_aliases": [],
      "id": "enhancer-promoter-interaction-prediction",
      "label": "Enhancer-promoter interaction prediction",
      "label_zh": "增强子-启动子相互作用预测",
      "object_ids": [
        "dna"
      ],
      "official_work_count": 1,
      "parent_id": "dna-regulation-perturbation",
      "task_family_id": "molecular-interaction"
    },
    {
      "aliases": [
        "chromatin mark prediction"
      ],
      "chromatin": null,
      "coverage_family_count": 2,
      "coverage_track_count": 1,
      "definition": "Predict DNA methylation",
      "deprecated_aliases": [],
      "histone": null,
      "id": "epigenetic-mark-prediction",
      "label": "Epigenetic-mark prediction",
      "label_zh": "表观遗传标记预测",
      "object_ids": [
        "dna"
      ],
      "official_work_count": 2,
      "or related epigenetic marks.": null,
      "parent_id": "dna-regulation-perturbation",
      "task_family_id": "sequence-regulation"
    },
    {
      "aliases": [
        "genomic variant effect"
      ],
      "coverage_family_count": 0,
      "coverage_track_count": 0,
      "definition": "Predict regulatory",
      "deprecated_aliases": [],
      "id": "dna-variant-effect-prediction",
      "label": "DNA variant-effect prediction",
      "label_zh": "DNA变异效应预测",
      "molecular": null,
      "object_ids": [
        "dna"
      ],
      "official_work_count": 0,
      "or phenotypic consequences of DNA variants.": null,
      "parent_id": "dna-regulation-perturbation",
      "task_family_id": "variant-perturbation"
    },
    {
      "aliases": [
        "DNA sequence reasoning",
        "restriction analysis",
        "PCR sequence analysis"
      ],
      "coverage_family_count": 2,
      "coverage_track_count": 5,
      "definition": "Analyze DNA sequence properties",
      "deprecated_aliases": [],
      "id": "dna-sequence-analysis",
      "label": "DNA sequence analysis",
      "label_zh": "DNA序列分析",
      "object_ids": [
        "dna"
      ],
      "official_work_count": 2,
      "open reading frames": null,
      "or amplicons when the evaluated task is not de novo sequence design.": null,
      "parent_id": "dna-regulation-perturbation",
      "primers": null,
      "restriction fragments": null,
      "task_family_id": "sequence-regulation"
    },
    {
      "aliases": [
        "CRISPR on-target prediction"
      ],
      "coverage_family_count": 2,
      "coverage_track_count": 1,
      "definition": "Predict on-target activity or efficiency of a CRISPR guide sequence.",
      "deprecated_aliases": [],
      "id": "crispr-guide-activity-prediction",
      "label": "CRISPR guide activity prediction",
      "label_zh": "CRISPR向导活性预测",
      "object_ids": [
        "dna",
        "rna"
      ],
      "official_work_count": 2,
      "parent_id": "dna-regulation-perturbation",
      "task_family_id": "variant-perturbation"
    },
    {
      "aliases": [
        "CRISPR off target",
        "guide off-target prediction"
      ],
      "coverage_family_count": 1,
      "coverage_track_count": 0,
      "definition": "Predict unintended CRISPR guide interactions or editing activity at off-target sequences.",
      "deprecated_aliases": [],
      "id": "crispr-off-target-prediction",
      "label": "CRISPR off-target prediction",
      "label_zh": "CRISPR脱靶预测",
      "object_ids": [
        "dna",
        "rna"
      ],
      "official_work_count": 1,
      "parent_id": "dna-regulation-perturbation",
      "task_family_id": "variant-perturbation"
    },
    {
      "aliases": [
        "regulatory sequence design"
      ],
      "coverage_family_count": 1,
      "coverage_track_count": 7,
      "definition": "Generate or optimize a DNA sequence for a specified regulatory or experimental function.",
      "deprecated_aliases": [],
      "id": "dna-sequence-design",
      "label": "DNA sequence design",
      "label_zh": "DNA序列设计",
      "object_ids": [
        "dna"
      ],
      "official_work_count": 2,
      "parent_id": "dna-regulation-perturbation",
      "task_family_id": "design-generation"
    },
    {
      "aliases": [
        "RNA tasks"
      ],
      "coverage_family_count": 0,
      "coverage_track_count": 0,
      "definition": "Predict or design RNA structure",
      "deprecated_aliases": [],
      "function": null,
      "id": "rna-function-design",
      "label": "RNA function and design",
      "label_zh": "RNA功能与设计",
      "object_ids": [
        "rna"
      ],
      "official_work_count": 0,
      "or intervention effects.": null,
      "parent_id": null,
      "processing": null,
      "task_family_id": "sequence-regulation",
      "translation": null
    },
    {
      "aliases": [
        "RNA folding"
      ],
      "coverage_family_count": 1,
      "coverage_track_count": 0,
      "definition": "Predict the secondary or tertiary structure of an RNA molecule.",
      "deprecated_aliases": [],
      "id": "rna-structure-prediction",
      "label": "RNA structure prediction",
      "label_zh": "RNA结构预测",
      "object_ids": [
        "rna"
      ],
      "official_work_count": 1,
      "parent_id": "rna-function-design",
      "task_family_id": "structure-prediction"
    },
    {
      "aliases": [
        "RNA structure ranking",
        "RNA model ranking"
      ],
      "coverage_family_count": 1,
      "coverage_track_count": 0,
      "definition": "Score or rank candidate RNA tertiary structures by their expected similarity to the native structure.",
      "deprecated_aliases": [],
      "id": "rna-structure-quality-assessment",
      "label": "RNA structure quality assessment",
      "label_zh": "RNA结构质量评估",
      "object_ids": [
        "rna"
      ],
      "official_work_count": 1,
      "parent_id": "rna-function-design",
      "task_family_id": "structure-prediction"
    },
    {
      "aliases": [
        "ncRNA classification"
      ],
      "coverage_family_count": 2,
      "coverage_track_count": 1,
      "definition": "Assign functional or biotype labels to an RNA sequence.",
      "deprecated_aliases": [],
      "id": "rna-function-classification",
      "label": "RNA function classification",
      "label_zh": "RNA功能分类",
      "object_ids": [
        "rna"
      ],
      "official_work_count": 2,
      "parent_id": "rna-function-design",
      "task_family_id": "property-function-prediction"
    },
    {
      "aliases": [
        "RNA modification site prediction"
      ],
      "coverage_family_count": 2,
      "coverage_track_count": 1,
      "definition": "Predict RNA modification types or sites from sequence or context.",
      "deprecated_aliases": [],
      "id": "rna-modification-prediction",
      "label": "RNA modification prediction",
      "label_zh": "RNA修饰预测",
      "object_ids": [
        "rna"
      ],
      "official_work_count": 2,
      "parent_id": "rna-function-design",
      "task_family_id": "sequence-regulation"
    },
    {
      "aliases": [
        "alternative polyadenylation",
        "ribosome loading",
        "translation efficiency"
      ],
      "coverage_family_count": 3,
      "coverage_track_count": 7,
      "definition": "Predict RNA processing",
      "deprecated_aliases": [],
      "id": "rna-processing-translation-prediction",
      "isoform usage": null,
      "label": "RNA processing and translation prediction",
      "label_zh": "RNA加工与翻译预测",
      "loading": null,
      "object_ids": [
        "rna"
      ],
      "official_work_count": 4,
      "or expression-related outcomes.": null,
      "parent_id": "rna-function-design",
      "task_family_id": "sequence-regulation",
      "translation": null
    },
    {
      "aliases": [
        "RNA degradation",
        "RNA stability"
      ],
      "coverage_family_count": 1,
      "coverage_track_count": 0,
      "definition": "Predict RNA stability",
      "degradation": null,
      "deprecated_aliases": [],
      "id": "rna-stability-degradation-prediction",
      "label": "RNA stability and degradation prediction",
      "label_zh": "RNA稳定性与降解预测",
      "object_ids": [
        "rna"
      ],
      "official_work_count": 1,
      "or condition-dependent decay measurements.": null,
      "parent_id": "rna-function-design",
      "task_family_id": "property-function-prediction"
    },
    {
      "aliases": [
        "RNA switch prediction",
        "programmable RNA switches"
      ],
      "coverage_family_count": 1,
      "coverage_track_count": 0,
      "definition": "Predict functional activity or expression measurements of engineered RNA switches.",
      "deprecated_aliases": [],
      "id": "rna-switch-activity-prediction",
      "label": "Programmable RNA-switch activity prediction",
      "label_zh": "可编程RNA开关活性预测",
      "object_ids": [
        "rna"
      ],
      "official_work_count": 1,
      "parent_id": "rna-function-design",
      "task_family_id": "property-function-prediction"
    },
    {
      "aliases": [
        "programmable RNA design",
        "RNA switch design"
      ],
      "coverage_family_count": 1,
      "coverage_track_count": 1,
      "definition": "Generate or optimize an RNA sequence for a specified functional response.",
      "deprecated_aliases": [],
      "id": "rna-design",
      "label": "RNA design",
      "label_zh": "RNA设计",
      "object_ids": [
        "rna"
      ],
      "official_work_count": 1,
      "parent_id": "rna-function-design",
      "task_family_id": "design-generation"
    },
    {
      "aliases": [
        "siRNA efficiency"
      ],
      "coverage_family_count": 1,
      "coverage_track_count": 1,
      "definition": "Predict the silencing efficacy of an siRNA sequence or sequence pair.",
      "deprecated_aliases": [],
      "id": "sirna-efficacy-prediction",
      "label": "siRNA efficacy prediction",
      "label_zh": "siRNA效能预测",
      "object_ids": [
        "rna"
      ],
      "official_work_count": 1,
      "parent_id": "rna-function-design",
      "task_family_id": "variant-perturbation"
    },
    {
      "aliases": [
        "drug discovery"
      ],
      "assess": null,
      "coverage_family_count": 0,
      "coverage_track_count": 0,
      "definition": "Generate",
      "deprecated_aliases": [],
      "id": "small-molecule-discovery",
      "label": "Small-molecule discovery",
      "label_zh": "小分子发现",
      "object_ids": [
        "small-molecule"
      ],
      "official_work_count": 0,
      "or plan the synthesis of small molecules for scientific or therapeutic use.": null,
      "parent_id": null,
      "task_family_id": "design-generation"
    },
    {
      "aliases": [
        "molecule generation",
        "de novo design"
      ],
      "coverage_family_count": 1,
      "coverage_track_count": 0,
      "definition": "Generate molecular structures satisfying requested constraints or objectives.",
      "deprecated_aliases": [],
      "id": "small-molecule-generation",
      "label": "Small-molecule generation",
      "label_zh": "小分子生成",
      "object_ids": [
        "small-molecule"
      ],
      "official_work_count": 1,
      "parent_id": "small-molecule-discovery",
      "task_family_id": "design-generation"
    },
    {
      "aliases": [
        "molecular property prediction",
        "QSAR"
      ],
      "coverage_family_count": 2,
      "coverage_track_count": 0,
      "definition": "Predict physicochemical or biological properties of a small molecule.",
      "deprecated_aliases": [],
      "id": "small-molecule-property-prediction",
      "label": "Small-molecule property prediction",
      "label_zh": "小分子性质预测",
      "object_ids": [
        "small-molecule"
      ],
      "official_work_count": 3,
      "parent_id": "small-molecule-discovery",
      "task_family_id": "property-function-prediction"
    },
    {
      "aliases": [
        "ADMET",
        "toxicity prediction"
      ],
      "coverage_family_count": 1,
      "coverage_track_count": 0,
      "definition": "Predict absorption",
      "deprecated_aliases": [],
      "distribution": null,
      "excretion": null,
      "id": "admet-toxicity-prediction",
      "label": "ADMET and toxicity prediction",
      "label_zh": "ADMET与毒性预测",
      "metabolism": null,
      "object_ids": [
        "small-molecule"
      ],
      "official_work_count": 2,
      "or toxicity outcomes.": null,
      "parent_id": "small-molecule-discovery",
      "safety": null,
      "task_family_id": "property-function-prediction"
    },
    {
      "aliases": [
        "reaction outcome prediction"
      ],
      "coverage_family_count": 0,
      "coverage_track_count": 0,
      "definition": "Predict products",
      "deprecated_aliases": [],
      "id": "reaction-prediction",
      "label": "Chemical reaction prediction",
      "label_zh": "化学反应预测",
      "object_ids": [
        "small-molecule"
      ],
      "official_work_count": 0,
      "or feasibility of a chemical reaction.": null,
      "parent_id": "small-molecule-discovery",
      "task_family_id": "property-function-prediction",
      "transformations": null
    },
    {
      "aliases": [
        "synthesis planning"
      ],
      "coverage_family_count": 0,
      "coverage_track_count": 0,
      "definition": "Propose precursor and reaction routes for synthesizing a target molecule.",
      "deprecated_aliases": [],
      "id": "retrosynthesis-planning",
      "label": "Retrosynthesis planning",
      "label_zh": "逆合成规划",
      "object_ids": [
        "small-molecule"
      ],
      "official_work_count": 0,
      "parent_id": "small-molecule-discovery",
      "task_family_id": "design-generation"
    },
    {
      "aliases": [
        "omics analysis"
      ],
      "cells": null,
      "coverage_family_count": 4,
      "coverage_track_count": 0,
      "definition": "Analyze genome-scale profiles",
      "deprecated_aliases": [],
      "id": "omics-cellular-analysis",
      "label": "Omics and cellular analysis",
      "label_zh": "组学与细胞分析",
      "object_ids": [
        "omics-profile",
        "cell",
        "tissue"
      ],
      "official_work_count": 7,
      "or integrated modalities.": null,
      "parent_id": null,
      "task_family_id": "omics-analysis",
      "tissues": null
    },
    {
      "aliases": [
        "DE analysis"
      ],
      "coverage_family_count": 1,
      "coverage_track_count": 0,
      "definition": "Identify genes or features with condition-associated expression changes.",
      "deprecated_aliases": [],
      "id": "differential-expression-analysis",
      "label": "Differential expression analysis",
      "label_zh": "差异表达分析",
      "object_ids": [
        "omics-profile"
      ],
      "official_work_count": 3,
      "parent_id": "omics-cellular-analysis",
      "task_family_id": "omics-analysis"
    },
    {
      "aliases": [
        "DA analysis"
      ],
      "coverage_family_count": 0,
      "coverage_track_count": 0,
      "definition": "Identify features",
      "deprecated_aliases": [],
      "id": "differential-abundance-analysis",
      "label": "Differential abundance analysis",
      "label_zh": "差异丰度分析",
      "object_ids": [
        "omics-profile",
        "microbial-community"
      ],
      "official_work_count": 0,
      "or metabolites with abundance changes.": null,
      "parent_id": "omics-cellular-analysis",
      "proteins": null,
      "task_family_id": "omics-analysis",
      "taxa": null
    },
    {
      "aliases": [
        "cell annotation"
      ],
      "coverage_family_count": 2,
      "coverage_track_count": 0,
      "definition": "Assign biological cell-type or state labels to single-cell profiles.",
      "deprecated_aliases": [],
      "id": "cell-type-annotation",
      "label": "Cell-type annotation",
      "label_zh": "细胞类型注释",
      "object_ids": [
        "cell",
        "omics-profile"
      ],
      "official_work_count": 4,
      "parent_id": "omics-cellular-analysis",
      "task_family_id": "cellular-spatial-analysis"
    },
    {
      "aliases": [
        "single-cell clustering"
      ],
      "coverage_family_count": 1,
      "coverage_track_count": 0,
      "definition": "Discover cell populations or states by clustering molecular profiles.",
      "deprecated_aliases": [],
      "id": "cell-state-clustering",
      "label": "Cell-state clustering",
      "label_zh": "细胞状态聚类",
      "object_ids": [
        "cell",
        "omics-profile"
      ],
      "official_work_count": 3,
      "parent_id": "omics-cellular-analysis",
      "task_family_id": "cellular-spatial-analysis"
    },
    {
      "aliases": [
        "single-cell batch integration",
        "batch correction",
        "scIB"
      ],
      "coverage_family_count": 1,
      "coverage_track_count": 0,
      "definition": "Integrate single-cell datasets across batches",
      "deprecated_aliases": [],
      "id": "single-cell-data-integration",
      "label": "Single-cell data integration",
      "label_zh": "单细胞数据整合",
      "object_ids": [
        "cell",
        "omics-profile"
      ],
      "official_work_count": 1,
      "or modalities while conserving biological variation.": null,
      "parent_id": "omics-cellular-analysis",
      "protocols": null,
      "studies": null,
      "task_family_id": "cellular-spatial-analysis"
    },
    {
      "aliases": [
        "pseudotime",
        "lineage inference"
      ],
      "coverage_family_count": 0,
      "coverage_track_count": 0,
      "definition": "Infer developmental",
      "deprecated_aliases": [],
      "id": "trajectory-inference",
      "label": "Trajectory inference",
      "label_zh": "轨迹推断",
      "object_ids": [
        "cell",
        "omics-profile"
      ],
      "official_work_count": 0,
      "or state-transition trajectories from cellular data.": null,
      "parent_id": "omics-cellular-analysis",
      "task_family_id": "cellular-spatial-analysis",
      "temporal": null
    },
    {
      "aliases": [
        "GRN inference"
      ],
      "coverage_family_count": 0,
      "coverage_track_count": 0,
      "definition": "Infer transcription-factor and target-gene regulatory relationships.",
      "deprecated_aliases": [],
      "id": "gene-regulatory-network-inference",
      "label": "Gene-regulatory-network inference",
      "label_zh": "基因调控网络推断",
      "object_ids": [
        "dna",
        "rna",
        "omics-profile"
      ],
      "official_work_count": 0,
      "parent_id": "omics-cellular-analysis",
      "task_family_id": "omics-analysis"
    },
    {
      "aliases": [
        "gene-set enrichment",
        "GSEA"
      ],
      "coverage_family_count": 0,
      "coverage_track_count": 0,
      "definition": "Identify pathways or gene sets enriched in a list or ranked molecular profile.",
      "deprecated_aliases": [],
      "id": "pathway-enrichment-analysis",
      "label": "Pathway enrichment analysis",
      "label_zh": "通路富集分析",
      "object_ids": [
        "omics-profile"
      ],
      "official_work_count": 0,
      "parent_id": "omics-cellular-analysis",
      "task_family_id": "omics-analysis"
    },
    {
      "aliases": [
        "spatial domain analysis",
        "spatial transcriptomics"
      ],
      "coverage_family_count": 1,
      "coverage_track_count": 0,
      "definition": "Analyze spatially resolved molecular profiles",
      "deprecated_aliases": [],
      "domains": null,
      "id": "spatial-omics-analysis",
      "label": "Spatial omics analysis",
      "label_zh": "空间组学分析",
      "neighborhoods": null,
      "object_ids": [
        "cell",
        "tissue",
        "omics-profile"
      ],
      "official_work_count": 3,
      "or tissue structure.": null,
      "parent_id": "omics-cellular-analysis",
      "task_family_id": "cellular-spatial-analysis"
    },
    {
      "aliases": [
        "multimodal integration"
      ],
      "coverage_family_count": 0,
      "coverage_track_count": 0,
      "definition": "Integrate two or more molecular modalities into a joint analysis or representation.",
      "deprecated_aliases": [],
      "id": "multiomics-integration",
      "label": "Multi-omics integration",
      "label_zh": "多组学整合",
      "object_ids": [
        "omics-profile"
      ],
      "official_work_count": 0,
      "parent_id": "omics-cellular-analysis",
      "task_family_id": "omics-analysis"
    },
    {
      "aliases": [
        "signature discovery"
      ],
      "coverage_family_count": 0,
      "coverage_track_count": 0,
      "definition": "Identify molecular features or signatures associated with phenotype",
      "deprecated_aliases": [],
      "diagnosis": null,
      "id": "biomarker-discovery",
      "label": "Biomarker discovery",
      "label_zh": "生物标志物发现",
      "object_ids": [
        "omics-profile",
        "organism-population"
      ],
      "official_work_count": 0,
      "or outcome.": null,
      "parent_id": "omics-cellular-analysis",
      "task_family_id": "omics-analysis"
    },
    {
      "aliases": [
        "microbiome analysis",
        "metagenomics"
      ],
      "coverage_family_count": 0,
      "coverage_track_count": 0,
      "definition": "Analyze microbial composition",
      "deprecated_aliases": [],
      "diversity": null,
      "function": null,
      "genomes": null,
      "id": "microbial-metagenomic-analysis",
      "label": "Microbial and metagenomic analysis",
      "label_zh": "微生物与宏基因组分析",
      "object_ids": [
        "microbial-community",
        "omics-profile"
      ],
      "official_work_count": 0,
      "or community variation.": null,
      "parent_id": "omics-cellular-analysis",
      "task_family_id": "omics-analysis"
    },
    {
      "aliases": [
        "statistical genetics"
      ],
      "ancestry": null,
      "coverage_family_count": 0,
      "coverage_track_count": 0,
      "definition": "Analyze genetic association",
      "deprecated_aliases": [],
      "id": "statistical-population-genetics-analysis",
      "inheritance": null,
      "label": "Statistical and population genetics",
      "label_zh": "统计与群体遗传分析",
      "object_ids": [
        "dna",
        "organism-population"
      ],
      "official_work_count": 0,
      "or population history.": null,
      "parent_id": null,
      "risk": null,
      "task_family_id": "statistical-genetics"
    },
    {
      "QTLs": null,
      "aliases": [
        "GWAS",
        "QTL mapping",
        "causal mapping"
      ],
      "coverage_family_count": 0,
      "coverage_track_count": 0,
      "definition": "Identify genetic associations",
      "deprecated_aliases": [],
      "id": "genetic-association-causal-mapping",
      "label": "Genetic association and causal mapping",
      "label_zh": "遗传关联与因果定位",
      "loci": null,
      "object_ids": [
        "dna",
        "organism-population"
      ],
      "official_work_count": 0,
      "or candidate causal variants.": null,
      "parent_id": "statistical-population-genetics-analysis",
      "task_family_id": "statistical-genetics"
    },
    {
      "aliases": [
        "heritability",
        "polygenic risk score",
        "PRS"
      ],
      "coverage_family_count": 0,
      "coverage_track_count": 0,
      "definition": "Estimate trait architecture",
      "deprecated_aliases": [],
      "heritability": null,
      "id": "heritability-polygenic-prediction",
      "label": "Heritability and polygenic prediction",
      "label_zh": "遗传力与多基因预测",
      "object_ids": [
        "dna",
        "organism-population"
      ],
      "official_work_count": 0,
      "or polygenic risk and prediction.": null,
      "parent_id": "statistical-population-genetics-analysis",
      "task_family_id": "statistical-genetics"
    },
    {
      "admixture": null,
      "aliases": [
        "ancestry inference",
        "demographic inference"
      ],
      "coverage_family_count": 0,
      "coverage_track_count": 0,
      "definition": "Analyze ancestry",
      "demography": null,
      "deprecated_aliases": [],
      "genealogies": null,
      "id": "population-genetics-analysis",
      "label": "Population genetics analysis",
      "label_zh": "群体遗传分析",
      "object_ids": [
        "dna",
        "organism-population"
      ],
      "official_work_count": 0,
      "or population structure.": null,
      "parent_id": "statistical-population-genetics-analysis",
      "selection": null,
      "task_family_id": "statistical-genetics"
    },
    {
      "aliases": [
        "scientific analysis workflow"
      ],
      "coverage_family_count": 0,
      "coverage_track_count": 0,
      "definition": "Retrieve evidence",
      "deprecated_aliases": [],
      "execute analyses": null,
      "id": "scientific-workflow-systems",
      "label": "Scientific workflow and systems analysis",
      "label_zh": "科学工作流与系统分析",
      "object_ids": [
        "experimental-system",
        "omics-profile"
      ],
      "official_work_count": 0,
      "or model dynamic systems.": null,
      "parent_id": null,
      "plan experiments": null,
      "task_family_id": "scientific-workflow"
    },
    {
      "aliases": [
        "database query",
        "record retrieval"
      ],
      "association": null,
      "coverage_family_count": 1,
      "coverage_track_count": 11,
      "definition": "Retrieve a scientific record",
      "deprecated_aliases": [],
      "id": "scientific-database-retrieval",
      "label": "Scientific database retrieval",
      "label_zh": "科学数据库检索",
      "object_ids": [
        "experimental-system"
      ],
      "official_work_count": 1,
      "or fact from a structured database.": null,
      "parent_id": "scientific-workflow-systems",
      "sequence": null,
      "task_family_id": "scientific-workflow"
    },
    {
      "aliases": [
        "figure interpretation",
        "table interpretation",
        "literature synthesis"
      ],
      "and supporting evidence.": null,
      "coverage_family_count": 2,
      "coverage_track_count": 5,
      "definition": "Interpret or reconcile scientific text",
      "deprecated_aliases": [],
      "figures": null,
      "id": "scientific-evidence-interpretation",
      "label": "Scientific evidence interpretation",
      "label_zh": "科学证据解读",
      "object_ids": [
        "experimental-system"
      ],
      "official_work_count": 4,
      "parent_id": "scientific-workflow-systems",
      "tables": null,
      "task_family_id": "scientific-workflow"
    },
    {
      "aliases": [
        "experimental design",
        "protocol design"
      ],
      "assay": null,
      "controls": null,
      "coverage_family_count": 1,
      "coverage_track_count": 2,
      "definition": "Design an experiment",
      "deprecated_aliases": [],
      "id": "experiment-protocol-planning",
      "label": "Experiment and protocol planning",
      "label_zh": "实验与方案规划",
      "object_ids": [
        "experimental-system"
      ],
      "official_work_count": 3,
      "or follow-up plan.": null,
      "parent_id": "scientific-workflow-systems",
      "protocol": null,
      "task_family_id": "scientific-workflow"
    },
    {
      "aliases": [
        "agentic bioinformatics",
        "computational investigation"
      ],
      "and interpretation.": null,
      "code": null,
      "coverage_family_count": 5,
      "coverage_track_count": 1,
      "definition": "Complete a multi-step scientific analysis using data",
      "deprecated_aliases": [],
      "id": "end-to-end-computational-analysis",
      "label": "End-to-end computational analysis",
      "label_zh": "端到端计算分析",
      "object_ids": [
        "omics-profile",
        "experimental-system"
      ],
      "official_work_count": 8,
      "parent_id": "scientific-workflow-systems",
      "task_family_id": "scientific-workflow",
      "tools": null
    },
    {
      "aliases": [
        "SBML reconstruction",
        "pathway model reconstruction"
      ],
      "coverage_family_count": 1,
      "coverage_track_count": 2,
      "definition": "Reconstruct a biochemical reaction network or executable systems model.",
      "deprecated_aliases": [],
      "id": "reaction-network-reconstruction",
      "label": "Reaction-network reconstruction",
      "label_zh": "反应网络重建",
      "object_ids": [
        "experimental-system"
      ],
      "official_work_count": 1,
      "parent_id": "scientific-workflow-systems",
      "task_family_id": "systems-modeling"
    },
    {
      "aliases": [
        "interactive simulation",
        "system identification"
      ],
      "coverage_family_count": 1,
      "coverage_track_count": 2,
      "definition": "Interact with or reason over a simulator to identify",
      "deprecated_aliases": [],
      "id": "simulation-based-experiment",
      "label": "Simulation-based experiment",
      "label_zh": "仿真实验",
      "object_ids": [
        "experimental-system"
      ],
      "official_work_count": 1,
      "or characterize a biological system.": null,
      "parent_id": "scientific-workflow-systems",
      "task_family_id": "systems-modeling",
      "test": null
    }
  ],
  "taxonomies": {
    "access_levels": [
      {
        "definition": "Tasks and required evaluation materials are publicly accessible.",
        "deprecated_aliases": [],
        "id": "fully-open",
        "label": "Fully open",
        "label_zh": "完全公开",
        "parent_id": null
      },
      {
        "artifacts": null,
        "definition": "Some tasks",
        "deprecated_aliases": [],
        "id": "partially-open",
        "label": "Partially open",
        "label_zh": "部分公开",
        "labels": null,
        "or graders are restricted.": null,
        "parent_id": null
      },
      {
        "definition": "Public descriptions exist but runnable evaluation materials are unavailable.",
        "deprecated_aliases": [],
        "id": "metadata-only",
        "label": "Metadata only",
        "label_zh": "仅元数据",
        "parent_id": null
      },
      {
        "definition": "The benchmark is reported but not publicly released.",
        "deprecated_aliases": [
          "internal"
        ],
        "id": "private-internal",
        "label": "Private or internal",
        "label_zh": "私有或内部",
        "parent_id": null
      }
    ],
    "capabilities": [
      {
        "definition": "Recall and apply scientific knowledge.",
        "deprecated_aliases": [],
        "id": "knowledge",
        "label": "Knowledge",
        "label_zh": "知识",
        "parent_id": null
      },
      {
        "and records.": null,
        "definition": "Interpret and reconcile papers",
        "deprecated_aliases": [],
        "figures": null,
        "id": "evidence-synthesis",
        "label": "Evidence synthesis",
        "label_zh": "证据综合",
        "parent_id": null,
        "tables": null
      },
      {
        "definition": "Retrieve scientific records",
        "deprecated_aliases": [],
        "id": "retrieval",
        "label": "Retrieval",
        "label_zh": "检索",
        "or evidence.": null,
        "parent_id": null,
        "sequences": null
      },
      {
        "definition": "Predict scientific properties or outcomes.",
        "deprecated_aliases": [],
        "id": "prediction",
        "label": "Prediction",
        "label_zh": "预测",
        "parent_id": null
      },
      {
        "definition": "Assign discrete biological labels.",
        "deprecated_aliases": [],
        "id": "classification",
        "label": "Classification",
        "label_zh": "分类",
        "parent_id": "prediction"
      },
      {
        "definition": "Predict quantitative biological values.",
        "deprecated_aliases": [],
        "id": "regression",
        "label": "Regression",
        "label_zh": "回归",
        "parent_id": "prediction"
      },
      {
        "assays": null,
        "constructs": null,
        "definition": "Design molecules",
        "deprecated_aliases": [],
        "experiments": null,
        "id": "design",
        "label": "Design",
        "label_zh": "设计",
        "or protocols.": null,
        "parent_id": null
      },
      {
        "constructs": null,
        "definition": "Generate sequences",
        "deprecated_aliases": [],
        "id": "generation",
        "label": "Generation",
        "label_zh": "生成",
        "or plans.": null,
        "parent_id": "design",
        "structures": null
      },
      {
        "definition": "Improve a scientific object or process under constraints.",
        "deprecated_aliases": [],
        "id": "optimization",
        "label": "Optimization",
        "label_zh": "优化",
        "parent_id": "design"
      },
      {
        "definition": "Analyze quantitative",
        "deprecated_aliases": [],
        "id": "data-analysis",
        "label": "Data analysis",
        "label_zh": "数据分析",
        "or experimental data.": null,
        "parent_id": null,
        "statistical": null
      },
      {
        "definition": "Write and execute analysis code.",
        "deprecated_aliases": [],
        "id": "coding",
        "label": "Coding",
        "label_zh": "编程",
        "parent_id": "data-analysis"
      },
      {
        "browsers": null,
        "databases": null,
        "definition": "Use scientific software",
        "deprecated_aliases": [],
        "id": "tool-use",
        "label": "Tool use",
        "label_zh": "工具使用",
        "or lab interfaces.": null,
        "parent_id": null
      },
      {
        "and follow-up studies.": null,
        "controls": null,
        "definition": "Plan experiments",
        "deprecated_aliases": [],
        "id": "experiment-planning",
        "label": "Experiment planning",
        "label_zh": "实验规划",
        "parent_id": "design"
      },
      {
        "definition": "Diagnose experimental or analytical failures.",
        "deprecated_aliases": [],
        "id": "troubleshooting",
        "label": "Troubleshooting",
        "label_zh": "故障排查",
        "parent_id": null
      },
      {
        "and reason under uncertainty.": null,
        "definition": "Explain mechanisms",
        "deprecated_aliases": [],
        "id": "scientific-reasoning",
        "label": "Scientific reasoning",
        "label_zh": "科学推理",
        "parent_id": null,
        "test hypotheses": null
      },
      {
        "definition": "Produce expert-useful scientific explanations and reports.",
        "deprecated_aliases": [],
        "id": "scientific-communication",
        "label": "Scientific communication",
        "label_zh": "科学交流",
        "parent_id": null
      }
    ],
    "domains": [
      {
        "definition": "Broad applied and foundational life-science research.",
        "deprecated_aliases": [],
        "id": "life-science",
        "label": "Life science",
        "label_zh": "生命科学",
        "parent_id": null
      },
      {
        "and engineering.": null,
        "definition": "Protein sequence",
        "deprecated_aliases": [],
        "function": null,
        "id": "protein-science",
        "interaction": null,
        "label": "Protein science",
        "label_zh": "蛋白质科学",
        "parent_id": "life-science",
        "structure": null
      },
      {
        "definition": "Amino-acid sequence understanding and modeling.",
        "deprecated_aliases": [],
        "id": "protein-sequence",
        "label": "Protein sequence",
        "label_zh": "蛋白质序列",
        "parent_id": "protein-science"
      },
      {
        "and structural interpretation.": null,
        "complexes": null,
        "coordinates": null,
        "definition": "Protein folding",
        "deprecated_aliases": [],
        "id": "protein-structure",
        "label": "Protein structure",
        "label_zh": "蛋白质结构",
        "parent_id": "protein-science"
      },
      {
        "constructs": null,
        "definition": "Designing or optimizing protein sequences",
        "deprecated_aliases": [],
        "id": "protein-design",
        "label": "Protein design",
        "label_zh": "蛋白质设计",
        "or functions.": null,
        "parent_id": "protein-science",
        "structures": null
      },
      {
        "complex formation": null,
        "definition": "Protein-protein interaction",
        "deprecated_aliases": [],
        "id": "protein-protein-binding",
        "label": "Protein-protein binding",
        "label_zh": "蛋白-蛋白结合",
        "or binding prediction.": null,
        "parent_id": "protein-science"
      },
      {
        "definition": "Binding between proteins and small molecules or other ligands.",
        "deprecated_aliases": [],
        "id": "protein-ligand-binding",
        "label": "Protein-ligand binding",
        "label_zh": "蛋白-配体结合",
        "parent_id": "protein-science"
      },
      {
        "affinity": null,
        "definition": "Antibody specificity",
        "deprecated_aliases": [],
        "epitope": null,
        "id": "antibody-antigen",
        "label": "Antibody-antigen",
        "label_zh": "抗体-抗原",
        "or antigen interaction.": null,
        "parent_id": "protein-protein-binding"
      },
      {
        "association": null,
        "definition": "Interaction",
        "deprecated_aliases": [
          "rna-protein-interaction"
        ],
        "id": "rna-protein-binding",
        "label": "RNA-protein binding",
        "label_zh": "RNA-蛋白结合",
        "or binding prediction between RNA and protein molecules.": null,
        "parent_id": "life-science"
      },
      {
        "and genetic mechanisms.": null,
        "definition": "Genome variation",
        "deprecated_aliases": [],
        "id": "genomics",
        "label": "Genomics",
        "label_zh": "基因组学",
        "parent_id": "life-science",
        "regulation": null,
        "sequencing": null
      },
      {
        "definition": "Bulk and other transcriptome-level measurements and analyses.",
        "deprecated_aliases": [],
        "id": "transcriptomics",
        "label": "Transcriptomics",
        "label_zh": "转录组学",
        "parent_id": "life-science"
      },
      {
        "and related assays.": null,
        "chromatin": null,
        "definition": "DNA methylation",
        "deprecated_aliases": [],
        "histone marks": null,
        "id": "epigenomics",
        "label": "Epigenomics",
        "label_zh": "表观组学",
        "parent_id": "life-science"
      },
      {
        "definition": "Single-cell molecular measurements and analyses.",
        "deprecated_aliases": [],
        "id": "single-cell",
        "label": "Single-cell",
        "label_zh": "单细胞组学",
        "parent_id": "life-science"
      },
      {
        "definition": "Spatially resolved molecular measurements and analyses.",
        "deprecated_aliases": [],
        "id": "spatial-omics",
        "label": "Spatial omics",
        "label_zh": "空间组学",
        "parent_id": "life-science"
      },
      {
        "definition": "Proteome-scale measurement and analysis.",
        "deprecated_aliases": [],
        "id": "proteomics",
        "label": "Proteomics",
        "label_zh": "蛋白质组学",
        "parent_id": "life-science"
      },
      {
        "definition": "Metabolite-scale measurement and analysis.",
        "deprecated_aliases": [],
        "id": "metabolomics",
        "label": "Metabolomics",
        "label_zh": "代谢组学",
        "parent_id": "life-science"
      },
      {
        "definition": "Microbial community composition and function.",
        "deprecated_aliases": [],
        "id": "microbiome",
        "label": "Microbiome",
        "label_zh": "微生物组",
        "parent_id": "life-science"
      },
      {
        "definition": "Integrated analysis across two or more omics modalities.",
        "deprecated_aliases": [
          "multi-omics"
        ],
        "id": "multiomics",
        "label": "Multi-omics",
        "label_zh": "多组学",
        "parent_id": "life-science"
      },
      {
        "and molecular experiments.": null,
        "definition": "Cellular mechanisms",
        "deprecated_aliases": [],
        "id": "molecular-cell-biology",
        "label": "Molecular and cell biology",
        "label_zh": "分子与细胞生物学",
        "parent_id": "life-science",
        "pathways": null,
        "perturbations": null
      },
      {
        "and troubleshooting.": null,
        "controls": null,
        "definition": "Assay design",
        "deprecated_aliases": [],
        "id": "assay-screening",
        "label": "Assays and screening",
        "label_zh": "实验与筛选",
        "parent_id": "life-science",
        "screening": null,
        "validation": null
      },
      {
        "definition": "Computational analysis of biological data and databases.",
        "deprecated_aliases": [],
        "id": "bioinformatics",
        "label": "Bioinformatics",
        "label_zh": "生物信息学",
        "parent_id": "life-science"
      },
      {
        "ADME": null,
        "and optimization.": null,
        "definition": "Small molecules",
        "deprecated_aliases": [],
        "id": "medchem",
        "label": "Medicinal chemistry",
        "label_zh": "药物化学",
        "parent_id": "life-science",
        "structure-activity relationships": null
      },
      {
        "and patient impact.": null,
        "biomarkers": null,
        "definition": "Preclinical-to-clinical reasoning",
        "deprecated_aliases": [],
        "id": "clinical-translational",
        "label": "Clinical and translational",
        "label_zh": "临床与转化",
        "parent_id": "life-science",
        "safety": null,
        "trials": null
      },
      {
        "and therapeutics.": null,
        "definition": "Viral sequences",
        "deprecated_aliases": [],
        "diagnostics": null,
        "evolution": null,
        "id": "virology",
        "label": "Virology",
        "label_zh": "病毒学",
        "parent_id": "life-science",
        "surveillance": null
      }
    ],
    "modalities": [
      {
        "definition": "Natural-language task context.",
        "deprecated_aliases": [],
        "id": "text",
        "label": "Text",
        "label_zh": "文本",
        "parent_id": null
      },
      {
        "definition": "Scientific papers",
        "deprecated_aliases": [],
        "id": "paper",
        "label": "Paper or document",
        "label_zh": "论文或文档",
        "or regulatory documents.": null,
        "parent_id": "text",
        "protocols": null,
        "reports": null
      },
      {
        "definition": "Tabular or spreadsheet data.",
        "deprecated_aliases": [],
        "id": "table",
        "label": "Table",
        "label_zh": "表格",
        "parent_id": null
      },
      {
        "definition": "Plots",
        "deprecated_aliases": [],
        "gels": null,
        "id": "figure",
        "label": "Figure",
        "label_zh": "科研图表",
        "microscopy": null,
        "or other scientific images.": null,
        "parent_id": null
      },
      {
        "definition": "Nucleotide sequences or sequence files.",
        "deprecated_aliases": [],
        "id": "dna-rna-sequence",
        "label": "DNA or RNA sequence",
        "label_zh": "DNA/RNA序列",
        "parent_id": null
      },
      {
        "definition": "Amino-acid sequences or sequence files.",
        "deprecated_aliases": [],
        "id": "protein-sequence",
        "label": "Protein sequence",
        "label_zh": "蛋白质序列",
        "parent_id": null
      },
      {
        "definition": "Small-molecule graphs, SMILES strings, fingerprints, or molecular structure records.",
        "deprecated_aliases": [
          "molecular graph",
          "SMILES"
        ],
        "id": "small-molecule-structure",
        "label": "Small-molecule structure",
        "label_zh": "小分子结构",
        "parent_id": null
      },
      {
        "complexes": null,
        "definition": "Molecular coordinates",
        "deprecated_aliases": [],
        "id": "structure-3d",
        "label": "3D structure",
        "label_zh": "三维结构",
        "or structure files.": null,
        "parent_id": null
      },
      {
        "definition": "Raw or minimally processed omics data.",
        "deprecated_aliases": [],
        "id": "raw-omics",
        "label": "Raw omics",
        "label_zh": "原始组学数据",
        "parent_id": null
      },
      {
        "definition": "General scientific image input.",
        "deprecated_aliases": [],
        "id": "image",
        "label": "Image",
        "label_zh": "图像",
        "parent_id": null
      },
      {
        "definition": "Structured external scientific databases.",
        "deprecated_aliases": [],
        "id": "database",
        "label": "Database",
        "label_zh": "数据库",
        "parent_id": null
      },
      {
        "definition": "Browser or URL-provided evidence.",
        "deprecated_aliases": [],
        "id": "web",
        "label": "Web",
        "label_zh": "网页",
        "parent_id": null
      },
      {
        "definition": "Source code",
        "deprecated_aliases": [],
        "id": "code",
        "label": "Code",
        "label_zh": "代码",
        "notebooks": null,
        "or executable analyses.": null,
        "parent_id": null
      },
      {
        "definition": "Instrument output or laboratory measurements.",
        "deprecated_aliases": [],
        "id": "wet-lab-output",
        "label": "Wet-lab output",
        "label_zh": "湿实验输出",
        "parent_id": null
      }
    ],
    "scientific_objects": [
      {
        "and measured properties.": null,
        "complexes": null,
        "definition": "Protein sequences",
        "deprecated_aliases": [],
        "functions": null,
        "id": "protein",
        "label": "Protein",
        "label_zh": "蛋白质",
        "parent_id": null,
        "structures": null
      },
      {
        "and regulatory elements.": null,
        "definition": "DNA sequences",
        "deprecated_aliases": [],
        "genomic variants": null,
        "id": "dna",
        "label": "DNA",
        "label_zh": "DNA",
        "parent_id": null
      },
      {
        "and regulation.": null,
        "definition": "RNA sequences",
        "deprecated_aliases": [],
        "id": "rna",
        "label": "RNA",
        "label_zh": "RNA",
        "parent_id": null,
        "processing": null,
        "structures": null,
        "translation": null
      },
      {
        "and other small molecular entities.": null,
        "definition": "Ligands",
        "deprecated_aliases": [
          "ligand"
        ],
        "drugs": null,
        "id": "small-molecule",
        "label": "Small molecule",
        "label_zh": "小分子",
        "metabolites": null,
        "parent_id": null
      },
      {
        "and populations.": null,
        "cell states": null,
        "definition": "Individual cells",
        "deprecated_aliases": [],
        "id": "cell",
        "label": "Cell",
        "label_zh": "细胞",
        "parent_id": null,
        "types": null
      },
      {
        "and spatially organized cellular systems.": null,
        "definition": "Tissues",
        "deprecated_aliases": [],
        "id": "tissue",
        "label": "Tissue",
        "label_zh": "组织",
        "organs": null,
        "parent_id": null
      },
      {
        "and biological populations.": null,
        "cohorts": null,
        "definition": "Organisms",
        "deprecated_aliases": [],
        "id": "organism-population",
        "label": "Organism or population",
        "label_zh": "个体或群体",
        "parent_id": null,
        "pedigrees": null
      },
      {
        "and microbiomes.": null,
        "definition": "Microbial communities",
        "deprecated_aliases": [],
        "id": "microbial-community",
        "label": "Microbial community",
        "label_zh": "微生物群落",
        "metagenomes": null,
        "parent_id": null
      },
      {
        "definition": "Genome-scale molecular measurement matrices and integrated omics profiles.",
        "deprecated_aliases": [],
        "id": "omics-profile",
        "label": "Omics profile",
        "label_zh": "组学谱",
        "parent_id": null
      },
      {
        "and experimental environments.": null,
        "definition": "Assays",
        "deprecated_aliases": [],
        "id": "experimental-system",
        "label": "Experimental system",
        "label_zh": "实验系统",
        "parent_id": null,
        "protocols": null,
        "reaction networks": null,
        "simulators": null
      }
    ],
    "scientific_tasks": [
      {
        "aliases": [
          "protein structure"
        ],
        "definition": "Predict or assess the three-dimensional structure of proteins or protein assemblies.",
        "deprecated_aliases": [],
        "id": "protein-structure",
        "label": "Protein structure prediction",
        "label_zh": "蛋白质结构预测",
        "object_ids": [
          "protein"
        ],
        "parent_id": null,
        "task_family_id": "structure-prediction"
      },
      {
        "aliases": [
          "protein folding",
          "folding",
          "monomer prediction"
        ],
        "definition": "Predict the three-dimensional structure of a single protein chain or domain.",
        "deprecated_aliases": [],
        "id": "protein-monomer-structure-prediction",
        "label": "Protein monomer structure prediction",
        "label_zh": "蛋白质单体折叠与结构预测",
        "object_ids": [
          "protein"
        ],
        "parent_id": "protein-structure",
        "task_family_id": "structure-prediction"
      },
      {
        "aliases": [
          "complex prediction",
          "multimer prediction",
          "protein docking"
        ],
        "definition": "Predict the assembled structure or stoichiometry of a protein complex.",
        "deprecated_aliases": [],
        "id": "protein-complex-structure-prediction",
        "label": "Protein complex structure prediction",
        "label_zh": "蛋白质复合物结构预测",
        "object_ids": [
          "protein"
        ],
        "parent_id": "protein-structure",
        "task_family_id": "structure-prediction"
      },
      {
        "aliases": [
          "model accuracy estimation",
          "structure quality assessment"
        ],
        "definition": "Estimate the accuracy or quality of a predicted protein structure model.",
        "deprecated_aliases": [],
        "id": "protein-model-quality-assessment",
        "label": "Protein model quality assessment",
        "label_zh": "蛋白质模型质量评估",
        "object_ids": [
          "protein"
        ],
        "parent_id": "protein-structure",
        "task_family_id": "structure-prediction"
      },
      {
        "aliases": [
          "secondary structure prediction",
          "SSP"
        ],
        "coil": null,
        "definition": "Predict per-residue helix",
        "deprecated_aliases": [],
        "id": "protein-secondary-structure-prediction",
        "label": "Protein secondary-structure prediction",
        "label_zh": "蛋白质二级结构预测",
        "object_ids": [
          "protein"
        ],
        "or other secondary-structure labels from a protein sequence.": null,
        "parent_id": "protein-structure",
        "strand": null,
        "task_family_id": "structure-prediction"
      },
      {
        "aliases": [
          "contact prediction",
          "residue contact prediction"
        ],
        "definition": "Predict which residue pairs in a protein are spatially in contact.",
        "deprecated_aliases": [],
        "id": "protein-contact-map-prediction",
        "label": "Protein contact-map prediction",
        "label_zh": "蛋白质接触图预测",
        "object_ids": [
          "protein"
        ],
        "parent_id": "protein-structure",
        "task_family_id": "structure-prediction"
      },
      {
        "aliases": [
          "protein engineering"
        ],
        "definition": "Generate or optimize proteins under structural",
        "deprecated_aliases": [],
        "functional": null,
        "id": "protein-design",
        "label": "Protein design",
        "label_zh": "蛋白质设计",
        "object_ids": [
          "protein"
        ],
        "or fitness constraints.": null,
        "parent_id": null,
        "task_family_id": "design-generation"
      },
      {
        "aliases": [
          "sequence generation",
          "inverse folding"
        ],
        "definition": "Generate an amino-acid sequence satisfying specified structural or functional constraints.",
        "deprecated_aliases": [],
        "id": "protein-sequence-design",
        "label": "Protein sequence design",
        "label_zh": "蛋白质序列设计",
        "object_ids": [
          "protein"
        ],
        "parent_id": "protein-design",
        "task_family_id": "design-generation"
      },
      {
        "aliases": [
          "fitness optimization",
          "directed evolution design"
        ],
        "definition": "Select or optimize protein variants toward improved measured fitness.",
        "deprecated_aliases": [],
        "id": "protein-fitness-optimization",
        "label": "Protein fitness optimization",
        "label_zh": "蛋白质适应度优化",
        "object_ids": [
          "protein"
        ],
        "parent_id": "protein-design",
        "task_family_id": "design-generation"
      },
      {
        "aliases": [
          "protein property prediction"
        ],
        "definition": "Predict protein function",
        "deprecated_aliases": [],
        "fitness": null,
        "id": "protein-property-function",
        "label": "Protein property and function prediction",
        "label_zh": "蛋白质性质与功能预测",
        "object_ids": [
          "protein"
        ],
        "or clinically relevant effects.": null,
        "parent_id": null,
        "solubility": null,
        "stability": null,
        "task_family_id": "property-function-prediction"
      },
      {
        "aliases": [
          "enzyme function prediction",
          "EC prediction"
        ],
        "definition": "Assign molecular function",
        "deprecated_aliases": [],
        "enzyme class": null,
        "id": "protein-function-annotation",
        "label": "Protein function annotation",
        "label_zh": "蛋白质功能注释",
        "object_ids": [
          "protein"
        ],
        "or functional labels to a protein.": null,
        "parent_id": "protein-property-function",
        "task_family_id": "property-function-prediction"
      },
      {
        "aliases": [
          "remote homology",
          "fold recognition"
        ],
        "definition": "Detect structural or evolutionary relationships between distantly related protein sequences.",
        "deprecated_aliases": [],
        "id": "protein-remote-homology-detection",
        "label": "Protein remote-homology detection",
        "label_zh": "蛋白质远缘同源检测",
        "object_ids": [
          "protein"
        ],
        "parent_id": "protein-property-function",
        "task_family_id": "property-function-prediction"
      },
      {
        "aliases": [
          "residue identity",
          "amino-acid environment classification"
        ],
        "definition": "Predict the amino-acid identity associated with a local protein structural environment.",
        "deprecated_aliases": [],
        "id": "protein-residue-identity-prediction",
        "label": "Protein residue-identity prediction",
        "label_zh": "蛋白质残基身份预测",
        "object_ids": [
          "protein"
        ],
        "parent_id": "protein-property-function",
        "task_family_id": "property-function-prediction"
      },
      {
        "aliases": [
          "fluorescence prediction"
        ],
        "definition": "Predict a measured fluorescence phenotype from a protein sequence.",
        "deprecated_aliases": [],
        "id": "protein-fluorescence-prediction",
        "label": "Protein fluorescence prediction",
        "label_zh": "蛋白质荧光性质预测",
        "object_ids": [
          "protein"
        ],
        "parent_id": "protein-property-function",
        "task_family_id": "property-function-prediction"
      },
      {
        "aliases": [
          "variant effect prediction",
          "mutation effect"
        ],
        "definition": "Predict the functional or phenotypic effect of protein sequence variants.",
        "deprecated_aliases": [],
        "id": "protein-mutation-effect-prediction",
        "label": "Protein mutation-effect prediction",
        "label_zh": "蛋白质突变效应预测",
        "object_ids": [
          "protein"
        ],
        "parent_id": "protein-property-function",
        "task_family_id": "variant-perturbation"
      },
      {
        "aliases": [
          "fitness landscape prediction",
          "sequence-to-fitness"
        ],
        "definition": "Predict a quantitative or ranked protein fitness measurement from sequence.",
        "deprecated_aliases": [],
        "id": "protein-fitness-prediction",
        "label": "Protein fitness prediction",
        "label_zh": "蛋白质适应度预测",
        "object_ids": [
          "protein"
        ],
        "parent_id": "protein-property-function",
        "task_family_id": "property-function-prediction"
      },
      {
        "aliases": [
          "thermostability prediction",
          "melting temperature prediction"
        ],
        "definition": "Predict stability",
        "deprecated_aliases": [],
        "id": "protein-stability-prediction",
        "label": "Protein stability prediction",
        "label_zh": "蛋白质稳定性预测",
        "object_ids": [
          "protein"
        ],
        "or melting behavior of a protein.": null,
        "parent_id": "protein-property-function",
        "task_family_id": "property-function-prediction",
        "thermostability": null
      },
      {
        "aliases": [
          "solubility"
        ],
        "definition": "Predict whether or how strongly a protein is soluble under an assay condition.",
        "deprecated_aliases": [],
        "id": "protein-solubility-prediction",
        "label": "Protein solubility prediction",
        "label_zh": "蛋白质溶解性预测",
        "object_ids": [
          "protein"
        ],
        "parent_id": "protein-property-function",
        "task_family_id": "property-function-prediction"
      },
      {
        "aliases": [
          "variant pathogenicity",
          "clinical variant classification"
        ],
        "definition": "Classify or prioritize protein variants for clinical relevance or pathogenicity.",
        "deprecated_aliases": [],
        "id": "protein-clinical-variant-interpretation",
        "label": "Protein clinical variant interpretation",
        "label_zh": "蛋白质临床变异解读",
        "object_ids": [
          "protein"
        ],
        "parent_id": "protein-property-function",
        "task_family_id": "variant-perturbation"
      },
      {
        "aliases": [
          "molecular binding"
        ],
        "binding geometry": null,
        "definition": "Predict or analyze interactions",
        "deprecated_aliases": [],
        "id": "molecular-interaction-analysis",
        "label": "Molecular interaction and binding",
        "label_zh": "分子相互作用与结合",
        "object_ids": [
          "protein",
          "rna",
          "small-molecule"
        ],
        "or affinity between molecules.": null,
        "parent_id": null,
        "specificity": null,
        "task_family_id": "molecular-interaction"
      },
      {
        "aliases": [
          "PPI",
          "protein binding protein"
        ],
        "definition": "Predict whether proteins interact or identify an interaction partner or interface.",
        "deprecated_aliases": [],
        "id": "protein-protein-interaction-prediction",
        "label": "Protein-protein interaction prediction",
        "label_zh": "蛋白质-蛋白质相互作用预测",
        "object_ids": [
          "protein"
        ],
        "parent_id": "molecular-interaction-analysis",
        "task_family_id": "molecular-interaction"
      },
      {
        "aliases": [
          "protein interface prediction",
          "interface contact prediction"
        ],
        "definition": "Predict which residues from two proteins will contact one another when the proteins bind.",
        "deprecated_aliases": [],
        "id": "protein-protein-interface-prediction",
        "label": "Protein-protein interface prediction",
        "label_zh": "蛋白质-蛋白质界面预测",
        "object_ids": [
          "protein"
        ],
        "parent_id": "molecular-interaction-analysis",
        "task_family_id": "molecular-interaction"
      },
      {
        "aliases": [
          "mutation stability prediction",
          "complex stability change"
        ],
        "definition": "Predict whether a mutation increases or decreases the stability of a protein complex or interaction.",
        "deprecated_aliases": [],
        "id": "protein-complex-mutation-stability-prediction",
        "label": "Protein-complex mutation stability prediction",
        "label_zh": "蛋白质复合物突变稳定性预测",
        "object_ids": [
          "protein"
        ],
        "parent_id": "molecular-interaction-analysis",
        "task_family_id": "variant-perturbation"
      },
      {
        "aliases": [
          "PPI affinity"
        ],
        "definition": "Predict or rank the strength of binding between protein partners.",
        "deprecated_aliases": [],
        "id": "protein-protein-binding-affinity",
        "label": "Protein-protein binding affinity",
        "label_zh": "蛋白质-蛋白质结合亲和力",
        "object_ids": [
          "protein"
        ],
        "parent_id": "molecular-interaction-analysis",
        "task_family_id": "molecular-interaction"
      },
      {
        "aliases": [
          "antibody binding",
          "antigen binding",
          "neutralization"
        ],
        "definition": "Predict antibody-antigen recognition",
        "deprecated_aliases": [],
        "id": "antibody-antigen-interaction",
        "label": "Antibody-antigen interaction",
        "label_zh": "抗体-抗原相互作用",
        "neutralization": null,
        "object_ids": [
          "protein"
        ],
        "or affinity.": null,
        "parent_id": "molecular-interaction-analysis",
        "specificity": null,
        "structure": null,
        "task_family_id": "molecular-interaction"
      },
      {
        "aliases": [
          "protein ligand binding",
          "protein-small-molecule binding",
          "ligand binding"
        ],
        "definition": "Predict or reason about protein-ligand binding when the official source does not distinguish pose from affinity.",
        "deprecated_aliases": [],
        "id": "protein-ligand-binding-prediction",
        "label": "Protein-ligand binding prediction",
        "label_zh": "蛋白质-配体结合预测",
        "object_ids": [
          "protein",
          "small-molecule"
        ],
        "parent_id": "molecular-interaction-analysis",
        "task_family_id": "molecular-interaction"
      },
      {
        "aliases": [
          "ligand efficacy prediction",
          "ligand activation prediction"
        ],
        "definition": "Given a ligand and target structures, predict whether the ligand activates or inhibits the protein's function rather than whether it binds.",
        "deprecated_aliases": [],
        "id": "protein-ligand-efficacy-prediction",
        "label": "Protein-ligand functional efficacy prediction",
        "label_zh": "蛋白质-配体功能效应预测",
        "object_ids": [
          "protein",
          "small-molecule"
        ],
        "parent_id": "molecular-interaction-analysis",
        "task_family_id": "property-function-prediction"
      },
      {
        "aliases": [
          "molecular docking",
          "ligand docking",
          "pose prediction"
        ],
        "definition": "Predict the bound pose or geometry of a small molecule in a protein complex.",
        "deprecated_aliases": [],
        "id": "protein-ligand-pose-prediction",
        "label": "Protein-ligand pose prediction",
        "label_zh": "蛋白质-配体结合构象预测",
        "object_ids": [
          "protein",
          "small-molecule"
        ],
        "parent_id": "molecular-interaction-analysis",
        "task_family_id": "molecular-interaction"
      },
      {
        "aliases": [
          "ligand affinity",
          "binding affinity"
        ],
        "definition": "Predict or rank the strength of protein-small-molecule binding.",
        "deprecated_aliases": [],
        "id": "protein-ligand-binding-affinity",
        "label": "Protein-ligand binding affinity",
        "label_zh": "蛋白质-配体结合亲和力",
        "object_ids": [
          "protein",
          "small-molecule"
        ],
        "parent_id": "molecular-interaction-analysis",
        "task_family_id": "molecular-interaction"
      },
      {
        "aliases": [
          "RPI",
          "RNA protein binding"
        ],
        "definition": "Predict whether an RNA and protein interact or bind.",
        "deprecated_aliases": [],
        "id": "rna-protein-interaction-prediction",
        "label": "RNA-protein interaction prediction",
        "label_zh": "RNA-蛋白质相互作用预测",
        "object_ids": [
          "rna",
          "protein"
        ],
        "parent_id": "molecular-interaction-analysis",
        "task_family_id": "molecular-interaction"
      },
      {
        "aliases": [
          "regulatory genomics"
        ],
        "chromatin signals": null,
        "definition": "Predict DNA regulatory elements",
        "deprecated_aliases": [],
        "id": "dna-regulation-perturbation",
        "label": "DNA regulation and perturbation",
        "label_zh": "DNA调控与扰动",
        "object_ids": [
          "dna"
        ],
        "or editing outcomes.": null,
        "parent_id": null,
        "task_family_id": "sequence-regulation",
        "variants": null
      },
      {
        "aliases": [
          "core promoter prediction"
        ],
        "definition": "Identify promoter regions or promoter activity from DNA sequence.",
        "deprecated_aliases": [],
        "id": "promoter-detection",
        "label": "Promoter detection",
        "label_zh": "启动子识别",
        "object_ids": [
          "dna"
        ],
        "parent_id": "dna-regulation-perturbation",
        "task_family_id": "sequence-regulation"
      },
      {
        "aliases": [
          "enhancer prediction"
        ],
        "definition": "Predict enhancer activity or regulatory potential from DNA sequence.",
        "deprecated_aliases": [],
        "id": "enhancer-activity-prediction",
        "label": "Enhancer activity prediction",
        "label_zh": "增强子活性预测",
        "object_ids": [
          "dna"
        ],
        "parent_id": "dna-regulation-perturbation",
        "task_family_id": "sequence-regulation"
      },
      {
        "aliases": [
          "TFBS prediction"
        ],
        "definition": "Predict transcription-factor binding sites or occupancy from sequence or genomic context.",
        "deprecated_aliases": [],
        "id": "transcription-factor-binding-site-prediction",
        "label": "Transcription-factor binding-site prediction",
        "label_zh": "转录因子结合位点预测",
        "object_ids": [
          "dna",
          "protein"
        ],
        "parent_id": "dna-regulation-perturbation",
        "task_family_id": "molecular-interaction"
      },
      {
        "aliases": [
          "EPI prediction"
        ],
        "definition": "Predict physical or functional enhancer-promoter interactions.",
        "deprecated_aliases": [],
        "id": "enhancer-promoter-interaction-prediction",
        "label": "Enhancer-promoter interaction prediction",
        "label_zh": "增强子-启动子相互作用预测",
        "object_ids": [
          "dna"
        ],
        "parent_id": "dna-regulation-perturbation",
        "task_family_id": "molecular-interaction"
      },
      {
        "aliases": [
          "chromatin mark prediction"
        ],
        "chromatin": null,
        "definition": "Predict DNA methylation",
        "deprecated_aliases": [],
        "histone": null,
        "id": "epigenetic-mark-prediction",
        "label": "Epigenetic-mark prediction",
        "label_zh": "表观遗传标记预测",
        "object_ids": [
          "dna"
        ],
        "or related epigenetic marks.": null,
        "parent_id": "dna-regulation-perturbation",
        "task_family_id": "sequence-regulation"
      },
      {
        "aliases": [
          "genomic variant effect"
        ],
        "definition": "Predict regulatory",
        "deprecated_aliases": [],
        "id": "dna-variant-effect-prediction",
        "label": "DNA variant-effect prediction",
        "label_zh": "DNA变异效应预测",
        "molecular": null,
        "object_ids": [
          "dna"
        ],
        "or phenotypic consequences of DNA variants.": null,
        "parent_id": "dna-regulation-perturbation",
        "task_family_id": "variant-perturbation"
      },
      {
        "aliases": [
          "DNA sequence reasoning",
          "restriction analysis",
          "PCR sequence analysis"
        ],
        "definition": "Analyze DNA sequence properties",
        "deprecated_aliases": [],
        "id": "dna-sequence-analysis",
        "label": "DNA sequence analysis",
        "label_zh": "DNA序列分析",
        "object_ids": [
          "dna"
        ],
        "open reading frames": null,
        "or amplicons when the evaluated task is not de novo sequence design.": null,
        "parent_id": "dna-regulation-perturbation",
        "primers": null,
        "restriction fragments": null,
        "task_family_id": "sequence-regulation"
      },
      {
        "aliases": [
          "CRISPR on-target prediction"
        ],
        "definition": "Predict on-target activity or efficiency of a CRISPR guide sequence.",
        "deprecated_aliases": [],
        "id": "crispr-guide-activity-prediction",
        "label": "CRISPR guide activity prediction",
        "label_zh": "CRISPR向导活性预测",
        "object_ids": [
          "dna",
          "rna"
        ],
        "parent_id": "dna-regulation-perturbation",
        "task_family_id": "variant-perturbation"
      },
      {
        "aliases": [
          "CRISPR off target",
          "guide off-target prediction"
        ],
        "definition": "Predict unintended CRISPR guide interactions or editing activity at off-target sequences.",
        "deprecated_aliases": [],
        "id": "crispr-off-target-prediction",
        "label": "CRISPR off-target prediction",
        "label_zh": "CRISPR脱靶预测",
        "object_ids": [
          "dna",
          "rna"
        ],
        "parent_id": "dna-regulation-perturbation",
        "task_family_id": "variant-perturbation"
      },
      {
        "aliases": [
          "regulatory sequence design"
        ],
        "definition": "Generate or optimize a DNA sequence for a specified regulatory or experimental function.",
        "deprecated_aliases": [],
        "id": "dna-sequence-design",
        "label": "DNA sequence design",
        "label_zh": "DNA序列设计",
        "object_ids": [
          "dna"
        ],
        "parent_id": "dna-regulation-perturbation",
        "task_family_id": "design-generation"
      },
      {
        "aliases": [
          "RNA tasks"
        ],
        "definition": "Predict or design RNA structure",
        "deprecated_aliases": [],
        "function": null,
        "id": "rna-function-design",
        "label": "RNA function and design",
        "label_zh": "RNA功能与设计",
        "object_ids": [
          "rna"
        ],
        "or intervention effects.": null,
        "parent_id": null,
        "processing": null,
        "task_family_id": "sequence-regulation",
        "translation": null
      },
      {
        "aliases": [
          "RNA folding"
        ],
        "definition": "Predict the secondary or tertiary structure of an RNA molecule.",
        "deprecated_aliases": [],
        "id": "rna-structure-prediction",
        "label": "RNA structure prediction",
        "label_zh": "RNA结构预测",
        "object_ids": [
          "rna"
        ],
        "parent_id": "rna-function-design",
        "task_family_id": "structure-prediction"
      },
      {
        "aliases": [
          "RNA structure ranking",
          "RNA model ranking"
        ],
        "definition": "Score or rank candidate RNA tertiary structures by their expected similarity to the native structure.",
        "deprecated_aliases": [],
        "id": "rna-structure-quality-assessment",
        "label": "RNA structure quality assessment",
        "label_zh": "RNA结构质量评估",
        "object_ids": [
          "rna"
        ],
        "parent_id": "rna-function-design",
        "task_family_id": "structure-prediction"
      },
      {
        "aliases": [
          "ncRNA classification"
        ],
        "definition": "Assign functional or biotype labels to an RNA sequence.",
        "deprecated_aliases": [],
        "id": "rna-function-classification",
        "label": "RNA function classification",
        "label_zh": "RNA功能分类",
        "object_ids": [
          "rna"
        ],
        "parent_id": "rna-function-design",
        "task_family_id": "property-function-prediction"
      },
      {
        "aliases": [
          "RNA modification site prediction"
        ],
        "definition": "Predict RNA modification types or sites from sequence or context.",
        "deprecated_aliases": [],
        "id": "rna-modification-prediction",
        "label": "RNA modification prediction",
        "label_zh": "RNA修饰预测",
        "object_ids": [
          "rna"
        ],
        "parent_id": "rna-function-design",
        "task_family_id": "sequence-regulation"
      },
      {
        "aliases": [
          "alternative polyadenylation",
          "ribosome loading",
          "translation efficiency"
        ],
        "definition": "Predict RNA processing",
        "deprecated_aliases": [],
        "id": "rna-processing-translation-prediction",
        "isoform usage": null,
        "label": "RNA processing and translation prediction",
        "label_zh": "RNA加工与翻译预测",
        "loading": null,
        "object_ids": [
          "rna"
        ],
        "or expression-related outcomes.": null,
        "parent_id": "rna-function-design",
        "task_family_id": "sequence-regulation",
        "translation": null
      },
      {
        "aliases": [
          "RNA degradation",
          "RNA stability"
        ],
        "definition": "Predict RNA stability",
        "degradation": null,
        "deprecated_aliases": [],
        "id": "rna-stability-degradation-prediction",
        "label": "RNA stability and degradation prediction",
        "label_zh": "RNA稳定性与降解预测",
        "object_ids": [
          "rna"
        ],
        "or condition-dependent decay measurements.": null,
        "parent_id": "rna-function-design",
        "task_family_id": "property-function-prediction"
      },
      {
        "aliases": [
          "RNA switch prediction",
          "programmable RNA switches"
        ],
        "definition": "Predict functional activity or expression measurements of engineered RNA switches.",
        "deprecated_aliases": [],
        "id": "rna-switch-activity-prediction",
        "label": "Programmable RNA-switch activity prediction",
        "label_zh": "可编程RNA开关活性预测",
        "object_ids": [
          "rna"
        ],
        "parent_id": "rna-function-design",
        "task_family_id": "property-function-prediction"
      },
      {
        "aliases": [
          "programmable RNA design",
          "RNA switch design"
        ],
        "definition": "Generate or optimize an RNA sequence for a specified functional response.",
        "deprecated_aliases": [],
        "id": "rna-design",
        "label": "RNA design",
        "label_zh": "RNA设计",
        "object_ids": [
          "rna"
        ],
        "parent_id": "rna-function-design",
        "task_family_id": "design-generation"
      },
      {
        "aliases": [
          "siRNA efficiency"
        ],
        "definition": "Predict the silencing efficacy of an siRNA sequence or sequence pair.",
        "deprecated_aliases": [],
        "id": "sirna-efficacy-prediction",
        "label": "siRNA efficacy prediction",
        "label_zh": "siRNA效能预测",
        "object_ids": [
          "rna"
        ],
        "parent_id": "rna-function-design",
        "task_family_id": "variant-perturbation"
      },
      {
        "aliases": [
          "drug discovery"
        ],
        "assess": null,
        "definition": "Generate",
        "deprecated_aliases": [],
        "id": "small-molecule-discovery",
        "label": "Small-molecule discovery",
        "label_zh": "小分子发现",
        "object_ids": [
          "small-molecule"
        ],
        "or plan the synthesis of small molecules for scientific or therapeutic use.": null,
        "parent_id": null,
        "task_family_id": "design-generation"
      },
      {
        "aliases": [
          "molecule generation",
          "de novo design"
        ],
        "definition": "Generate molecular structures satisfying requested constraints or objectives.",
        "deprecated_aliases": [],
        "id": "small-molecule-generation",
        "label": "Small-molecule generation",
        "label_zh": "小分子生成",
        "object_ids": [
          "small-molecule"
        ],
        "parent_id": "small-molecule-discovery",
        "task_family_id": "design-generation"
      },
      {
        "aliases": [
          "molecular property prediction",
          "QSAR"
        ],
        "definition": "Predict physicochemical or biological properties of a small molecule.",
        "deprecated_aliases": [],
        "id": "small-molecule-property-prediction",
        "label": "Small-molecule property prediction",
        "label_zh": "小分子性质预测",
        "object_ids": [
          "small-molecule"
        ],
        "parent_id": "small-molecule-discovery",
        "task_family_id": "property-function-prediction"
      },
      {
        "aliases": [
          "ADMET",
          "toxicity prediction"
        ],
        "definition": "Predict absorption",
        "deprecated_aliases": [],
        "distribution": null,
        "excretion": null,
        "id": "admet-toxicity-prediction",
        "label": "ADMET and toxicity prediction",
        "label_zh": "ADMET与毒性预测",
        "metabolism": null,
        "object_ids": [
          "small-molecule"
        ],
        "or toxicity outcomes.": null,
        "parent_id": "small-molecule-discovery",
        "safety": null,
        "task_family_id": "property-function-prediction"
      },
      {
        "aliases": [
          "reaction outcome prediction"
        ],
        "definition": "Predict products",
        "deprecated_aliases": [],
        "id": "reaction-prediction",
        "label": "Chemical reaction prediction",
        "label_zh": "化学反应预测",
        "object_ids": [
          "small-molecule"
        ],
        "or feasibility of a chemical reaction.": null,
        "parent_id": "small-molecule-discovery",
        "task_family_id": "property-function-prediction",
        "transformations": null
      },
      {
        "aliases": [
          "synthesis planning"
        ],
        "definition": "Propose precursor and reaction routes for synthesizing a target molecule.",
        "deprecated_aliases": [],
        "id": "retrosynthesis-planning",
        "label": "Retrosynthesis planning",
        "label_zh": "逆合成规划",
        "object_ids": [
          "small-molecule"
        ],
        "parent_id": "small-molecule-discovery",
        "task_family_id": "design-generation"
      },
      {
        "aliases": [
          "omics analysis"
        ],
        "cells": null,
        "definition": "Analyze genome-scale profiles",
        "deprecated_aliases": [],
        "id": "omics-cellular-analysis",
        "label": "Omics and cellular analysis",
        "label_zh": "组学与细胞分析",
        "object_ids": [
          "omics-profile",
          "cell",
          "tissue"
        ],
        "or integrated modalities.": null,
        "parent_id": null,
        "task_family_id": "omics-analysis",
        "tissues": null
      },
      {
        "aliases": [
          "DE analysis"
        ],
        "definition": "Identify genes or features with condition-associated expression changes.",
        "deprecated_aliases": [],
        "id": "differential-expression-analysis",
        "label": "Differential expression analysis",
        "label_zh": "差异表达分析",
        "object_ids": [
          "omics-profile"
        ],
        "parent_id": "omics-cellular-analysis",
        "task_family_id": "omics-analysis"
      },
      {
        "aliases": [
          "DA analysis"
        ],
        "definition": "Identify features",
        "deprecated_aliases": [],
        "id": "differential-abundance-analysis",
        "label": "Differential abundance analysis",
        "label_zh": "差异丰度分析",
        "object_ids": [
          "omics-profile",
          "microbial-community"
        ],
        "or metabolites with abundance changes.": null,
        "parent_id": "omics-cellular-analysis",
        "proteins": null,
        "task_family_id": "omics-analysis",
        "taxa": null
      },
      {
        "aliases": [
          "cell annotation"
        ],
        "definition": "Assign biological cell-type or state labels to single-cell profiles.",
        "deprecated_aliases": [],
        "id": "cell-type-annotation",
        "label": "Cell-type annotation",
        "label_zh": "细胞类型注释",
        "object_ids": [
          "cell",
          "omics-profile"
        ],
        "parent_id": "omics-cellular-analysis",
        "task_family_id": "cellular-spatial-analysis"
      },
      {
        "aliases": [
          "single-cell clustering"
        ],
        "definition": "Discover cell populations or states by clustering molecular profiles.",
        "deprecated_aliases": [],
        "id": "cell-state-clustering",
        "label": "Cell-state clustering",
        "label_zh": "细胞状态聚类",
        "object_ids": [
          "cell",
          "omics-profile"
        ],
        "parent_id": "omics-cellular-analysis",
        "task_family_id": "cellular-spatial-analysis"
      },
      {
        "aliases": [
          "single-cell batch integration",
          "batch correction",
          "scIB"
        ],
        "definition": "Integrate single-cell datasets across batches",
        "deprecated_aliases": [],
        "id": "single-cell-data-integration",
        "label": "Single-cell data integration",
        "label_zh": "单细胞数据整合",
        "object_ids": [
          "cell",
          "omics-profile"
        ],
        "or modalities while conserving biological variation.": null,
        "parent_id": "omics-cellular-analysis",
        "protocols": null,
        "studies": null,
        "task_family_id": "cellular-spatial-analysis"
      },
      {
        "aliases": [
          "pseudotime",
          "lineage inference"
        ],
        "definition": "Infer developmental",
        "deprecated_aliases": [],
        "id": "trajectory-inference",
        "label": "Trajectory inference",
        "label_zh": "轨迹推断",
        "object_ids": [
          "cell",
          "omics-profile"
        ],
        "or state-transition trajectories from cellular data.": null,
        "parent_id": "omics-cellular-analysis",
        "task_family_id": "cellular-spatial-analysis",
        "temporal": null
      },
      {
        "aliases": [
          "GRN inference"
        ],
        "definition": "Infer transcription-factor and target-gene regulatory relationships.",
        "deprecated_aliases": [],
        "id": "gene-regulatory-network-inference",
        "label": "Gene-regulatory-network inference",
        "label_zh": "基因调控网络推断",
        "object_ids": [
          "dna",
          "rna",
          "omics-profile"
        ],
        "parent_id": "omics-cellular-analysis",
        "task_family_id": "omics-analysis"
      },
      {
        "aliases": [
          "gene-set enrichment",
          "GSEA"
        ],
        "definition": "Identify pathways or gene sets enriched in a list or ranked molecular profile.",
        "deprecated_aliases": [],
        "id": "pathway-enrichment-analysis",
        "label": "Pathway enrichment analysis",
        "label_zh": "通路富集分析",
        "object_ids": [
          "omics-profile"
        ],
        "parent_id": "omics-cellular-analysis",
        "task_family_id": "omics-analysis"
      },
      {
        "aliases": [
          "spatial domain analysis",
          "spatial transcriptomics"
        ],
        "definition": "Analyze spatially resolved molecular profiles",
        "deprecated_aliases": [],
        "domains": null,
        "id": "spatial-omics-analysis",
        "label": "Spatial omics analysis",
        "label_zh": "空间组学分析",
        "neighborhoods": null,
        "object_ids": [
          "cell",
          "tissue",
          "omics-profile"
        ],
        "or tissue structure.": null,
        "parent_id": "omics-cellular-analysis",
        "task_family_id": "cellular-spatial-analysis"
      },
      {
        "aliases": [
          "multimodal integration"
        ],
        "definition": "Integrate two or more molecular modalities into a joint analysis or representation.",
        "deprecated_aliases": [],
        "id": "multiomics-integration",
        "label": "Multi-omics integration",
        "label_zh": "多组学整合",
        "object_ids": [
          "omics-profile"
        ],
        "parent_id": "omics-cellular-analysis",
        "task_family_id": "omics-analysis"
      },
      {
        "aliases": [
          "signature discovery"
        ],
        "definition": "Identify molecular features or signatures associated with phenotype",
        "deprecated_aliases": [],
        "diagnosis": null,
        "id": "biomarker-discovery",
        "label": "Biomarker discovery",
        "label_zh": "生物标志物发现",
        "object_ids": [
          "omics-profile",
          "organism-population"
        ],
        "or outcome.": null,
        "parent_id": "omics-cellular-analysis",
        "task_family_id": "omics-analysis"
      },
      {
        "aliases": [
          "microbiome analysis",
          "metagenomics"
        ],
        "definition": "Analyze microbial composition",
        "deprecated_aliases": [],
        "diversity": null,
        "function": null,
        "genomes": null,
        "id": "microbial-metagenomic-analysis",
        "label": "Microbial and metagenomic analysis",
        "label_zh": "微生物与宏基因组分析",
        "object_ids": [
          "microbial-community",
          "omics-profile"
        ],
        "or community variation.": null,
        "parent_id": "omics-cellular-analysis",
        "task_family_id": "omics-analysis"
      },
      {
        "aliases": [
          "statistical genetics"
        ],
        "ancestry": null,
        "definition": "Analyze genetic association",
        "deprecated_aliases": [],
        "id": "statistical-population-genetics-analysis",
        "inheritance": null,
        "label": "Statistical and population genetics",
        "label_zh": "统计与群体遗传分析",
        "object_ids": [
          "dna",
          "organism-population"
        ],
        "or population history.": null,
        "parent_id": null,
        "risk": null,
        "task_family_id": "statistical-genetics"
      },
      {
        "QTLs": null,
        "aliases": [
          "GWAS",
          "QTL mapping",
          "causal mapping"
        ],
        "definition": "Identify genetic associations",
        "deprecated_aliases": [],
        "id": "genetic-association-causal-mapping",
        "label": "Genetic association and causal mapping",
        "label_zh": "遗传关联与因果定位",
        "loci": null,
        "object_ids": [
          "dna",
          "organism-population"
        ],
        "or candidate causal variants.": null,
        "parent_id": "statistical-population-genetics-analysis",
        "task_family_id": "statistical-genetics"
      },
      {
        "aliases": [
          "heritability",
          "polygenic risk score",
          "PRS"
        ],
        "definition": "Estimate trait architecture",
        "deprecated_aliases": [],
        "heritability": null,
        "id": "heritability-polygenic-prediction",
        "label": "Heritability and polygenic prediction",
        "label_zh": "遗传力与多基因预测",
        "object_ids": [
          "dna",
          "organism-population"
        ],
        "or polygenic risk and prediction.": null,
        "parent_id": "statistical-population-genetics-analysis",
        "task_family_id": "statistical-genetics"
      },
      {
        "admixture": null,
        "aliases": [
          "ancestry inference",
          "demographic inference"
        ],
        "definition": "Analyze ancestry",
        "demography": null,
        "deprecated_aliases": [],
        "genealogies": null,
        "id": "population-genetics-analysis",
        "label": "Population genetics analysis",
        "label_zh": "群体遗传分析",
        "object_ids": [
          "dna",
          "organism-population"
        ],
        "or population structure.": null,
        "parent_id": "statistical-population-genetics-analysis",
        "selection": null,
        "task_family_id": "statistical-genetics"
      },
      {
        "aliases": [
          "scientific analysis workflow"
        ],
        "definition": "Retrieve evidence",
        "deprecated_aliases": [],
        "execute analyses": null,
        "id": "scientific-workflow-systems",
        "label": "Scientific workflow and systems analysis",
        "label_zh": "科学工作流与系统分析",
        "object_ids": [
          "experimental-system",
          "omics-profile"
        ],
        "or model dynamic systems.": null,
        "parent_id": null,
        "plan experiments": null,
        "task_family_id": "scientific-workflow"
      },
      {
        "aliases": [
          "database query",
          "record retrieval"
        ],
        "association": null,
        "definition": "Retrieve a scientific record",
        "deprecated_aliases": [],
        "id": "scientific-database-retrieval",
        "label": "Scientific database retrieval",
        "label_zh": "科学数据库检索",
        "object_ids": [
          "experimental-system"
        ],
        "or fact from a structured database.": null,
        "parent_id": "scientific-workflow-systems",
        "sequence": null,
        "task_family_id": "scientific-workflow"
      },
      {
        "aliases": [
          "figure interpretation",
          "table interpretation",
          "literature synthesis"
        ],
        "and supporting evidence.": null,
        "definition": "Interpret or reconcile scientific text",
        "deprecated_aliases": [],
        "figures": null,
        "id": "scientific-evidence-interpretation",
        "label": "Scientific evidence interpretation",
        "label_zh": "科学证据解读",
        "object_ids": [
          "experimental-system"
        ],
        "parent_id": "scientific-workflow-systems",
        "tables": null,
        "task_family_id": "scientific-workflow"
      },
      {
        "aliases": [
          "experimental design",
          "protocol design"
        ],
        "assay": null,
        "controls": null,
        "definition": "Design an experiment",
        "deprecated_aliases": [],
        "id": "experiment-protocol-planning",
        "label": "Experiment and protocol planning",
        "label_zh": "实验与方案规划",
        "object_ids": [
          "experimental-system"
        ],
        "or follow-up plan.": null,
        "parent_id": "scientific-workflow-systems",
        "protocol": null,
        "task_family_id": "scientific-workflow"
      },
      {
        "aliases": [
          "agentic bioinformatics",
          "computational investigation"
        ],
        "and interpretation.": null,
        "code": null,
        "definition": "Complete a multi-step scientific analysis using data",
        "deprecated_aliases": [],
        "id": "end-to-end-computational-analysis",
        "label": "End-to-end computational analysis",
        "label_zh": "端到端计算分析",
        "object_ids": [
          "omics-profile",
          "experimental-system"
        ],
        "parent_id": "scientific-workflow-systems",
        "task_family_id": "scientific-workflow",
        "tools": null
      },
      {
        "aliases": [
          "SBML reconstruction",
          "pathway model reconstruction"
        ],
        "definition": "Reconstruct a biochemical reaction network or executable systems model.",
        "deprecated_aliases": [],
        "id": "reaction-network-reconstruction",
        "label": "Reaction-network reconstruction",
        "label_zh": "反应网络重建",
        "object_ids": [
          "experimental-system"
        ],
        "parent_id": "scientific-workflow-systems",
        "task_family_id": "systems-modeling"
      },
      {
        "aliases": [
          "interactive simulation",
          "system identification"
        ],
        "definition": "Interact with or reason over a simulator to identify",
        "deprecated_aliases": [],
        "id": "simulation-based-experiment",
        "label": "Simulation-based experiment",
        "label_zh": "仿真实验",
        "object_ids": [
          "experimental-system"
        ],
        "or characterize a biological system.": null,
        "parent_id": "scientific-workflow-systems",
        "task_family_id": "systems-modeling",
        "test": null
      }
    ],
    "task_families": [
      {
        "complex": null,
        "definition": "Infer a molecular structure",
        "deprecated_aliases": [],
        "id": "structure-prediction",
        "label": "Structure prediction",
        "label_zh": "结构预测",
        "or model quality.": null,
        "parent_id": null,
        "pose": null
      },
      {
        "construct": null,
        "definition": "Generate or optimize a molecular sequence",
        "deprecated_aliases": [],
        "id": "design-generation",
        "label": "Design and generation",
        "label_zh": "设计与生成",
        "or intervention.": null,
        "parent_id": null,
        "structure": null
      },
      {
        "binding": null,
        "definition": "Predict interaction",
        "deprecated_aliases": [
          "binding"
        ],
        "id": "molecular-interaction",
        "interface": null,
        "label": "Molecular interaction",
        "label_zh": "分子相互作用",
        "or affinity between molecular partners.": null,
        "parent_id": null,
        "pose": null,
        "specificity": null
      },
      {
        "biological functions": null,
        "definition": "Predict molecular properties",
        "deprecated_aliases": [],
        "fitness": null,
        "id": "property-function-prediction",
        "label": "Property and function prediction",
        "label_zh": "性质与功能预测",
        "or phenotypes.": null,
        "parent_id": null
      },
      {
        "definition": "Identify functional sequence elements",
        "deprecated_aliases": [],
        "expression": null,
        "id": "sequence-regulation",
        "label": "Sequence and regulation",
        "label_zh": "序列与调控",
        "or regulatory activity.": null,
        "parent_id": null,
        "processing": null
      },
      {
        "definition": "Predict the consequence of a mutation",
        "deprecated_aliases": [],
        "edit": null,
        "id": "variant-perturbation",
        "intervention": null,
        "label": "Variant and perturbation effect",
        "label_zh": "变异与扰动效应",
        "or perturbation.": null,
        "parent_id": null
      },
      {
        "conditions": null,
        "definition": "Analyze molecular profiles across samples",
        "deprecated_aliases": [],
        "id": "omics-analysis",
        "label": "Omics analysis",
        "label_zh": "组学分析",
        "or modalities.": null,
        "parent_id": null
      },
      {
        "definition": "Analyze cell identity",
        "deprecated_aliases": [],
        "id": "cellular-spatial-analysis",
        "label": "Cellular and spatial analysis",
        "label_zh": "细胞与空间分析",
        "or spatial context.": null,
        "organization": null,
        "parent_id": null,
        "state": null,
        "trajectories": null
      },
      {
        "ancestry": null,
        "definition": "Analyze association",
        "deprecated_aliases": [],
        "id": "statistical-genetics",
        "inheritance": null,
        "label": "Statistical genetics",
        "label_zh": "统计遗传学",
        "or polygenic architecture.": null,
        "parent_id": null,
        "population history": null
      },
      {
        "definition": "Retrieve",
        "deprecated_aliases": [],
        "id": "scientific-workflow",
        "interpret": null,
        "label": "Scientific workflow",
        "label_zh": "科学工作流",
        "or execute an end-to-end scientific workflow.": null,
        "parent_id": null,
        "plan": null
      },
      {
        "definition": "Reconstruct",
        "deprecated_aliases": [],
        "id": "systems-modeling",
        "identify": null,
        "label": "Systems modeling",
        "label_zh": "系统建模",
        "or reason over dynamic biological systems.": null,
        "parent_id": null,
        "simulate": null
      }
    ]
  },
  "works": [
    {
      "arxiv": null,
      "authors": [
        "Sebastian Lukasiak",
        "Alex Kalinka",
        "Nikhil Gupta",
        "Angelos Papadopoulos",
        "Khalid Saeed",
        "Ultan McDermott",
        "Gregory James Hannon",
        "Douglas Ross-Thriepland",
        "David Walter"
      ],
      "benchmark_ids": [
        "benchmark-dual-human-crispr-cas9-library",
        "benchmark-human-crispr-cas9-library"
      ],
      "benchmark_use_ids": [
        "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-benchmark-dual-human-crispr-cas9-library-3-use",
        "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-benchmark-dual-human-crispr-cas9-library-4-use",
        "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-benchmark-human-crispr-cas9-library-1-use",
        "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-benchmark-human-crispr-cas9-library-2-use"
      ],
      "canonical_url": "https://doi.org/10.1186/s12864-025-11386-3",
      "current_version_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-pmc-version-1",
      "doi": "10.1186/s12864-025-11386-3",
      "entity_type": "work",
      "evaluation_run_ids": [],
      "id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg",
      "organizations": [
        "Joint Astrazeneca-Cancer Research Horizons Functional Genomics Centre"
      ],
      "publication_date": "2025-02-26",
      "review_provenance": {
        "codex_cli_version": "codex-cli 0.146.0-alpha.9.2",
        "execution_surface": "local-codex-cli",
        "extractor_model_requested": "gpt-5.6-sol",
        "extractor_model_resolved": null,
        "generated_at": "2026-08-02T12:44:02+00:00",
        "local_run_id": "e0b79cbc-5e78-41f6-a2ac-5352db58aec4",
        "method": "local-codex-double-pass",
        "model_resolution_status": "not-reported",
        "pipeline_version": "1.4.0",
        "prompt_version": "paper-evidence-local-v15",
        "source_version_id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-pmc-version-1",
        "verifier_model_requested": "gpt-5.6-sol",
        "verifier_model_resolved": null
      },
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": null,
          "canonical_url": "https://doi.org/10.1186/s12864-025-11386-3",
          "content_sha256": "35eadf11308ce68bbab41ecd842ec948338a5bb1bfb394bd6eb25741e6de068e",
          "content_type": "application/xml",
          "doi": "10.1186/s12864-025-11386-3",
          "id": "a-benchmark-comparison-of-crisprn-guide-rna-design-alg-pmc-version-1",
          "label": "PMC version 1",
          "publication_date": "2025-02-26",
          "retrieved_at": "2026-08-02T12:44:02+00:00",
          "source_access": "open-url",
          "status": "version-of-record"
        }
      ],
      "status": "published",
      "title": "A benchmark comparison of CRISPRn guide-RNA design algorithms and generation of small single and dual-targeting libraries to boost screening efficiency",
      "verification": {
        "last_verified": "2026-08-02",
        "notes": "AI-assisted double-pass extraction; production inclusion still requires the owner SHA-comment gate.",
        "status": "verified"
      },
      "work_type": "paper"
    },
    {
      "arxiv": null,
      "authors": [
        "Sarah Sirin",
        "James R. Apgar",
        "Eric M. Bennett",
        "Amy E. Keating"
      ],
      "benchmark_ids": [
        "ab-bind"
      ],
      "benchmark_use_ids": [
        "ab-bind-antibody-binding-mutational-database-for-compu-ab-bind-1-use",
        "ab-bind-antibody-binding-mutational-database-for-compu-ab-bind-2-use",
        "ab-bind-antibody-binding-mutational-database-for-compu-ab-bind-3-use"
      ],
      "canonical_url": "https://doi.org/10.1002/pro.2829",
      "current_version_id": "ab-bind-antibody-binding-mutational-database-for-compu-2015-11-06",
      "doi": "10.1002/pro.2829",
      "entity_type": "work",
      "evaluation_run_ids": [],
      "id": "ab-bind-antibody-binding-mutational-database-for-compu",
      "organizations": [
        "Massachusetts Institute of Technology",
        "Pfizer Inc."
      ],
      "publication_date": "2015-11-06",
      "review_provenance": {
        "codex_cli_version": "codex-cli 0.147.0-alpha.6.5",
        "execution_surface": "local-codex-cli",
        "extractor_model_requested": "gpt-5.6-sol",
        "extractor_model_resolved": null,
        "generated_at": "2026-08-14T05:40:07+00:00",
        "local_run_id": "7ffee2e8-83f2-4132-a04c-a14692ff8a97",
        "method": "local-codex-double-pass",
        "model_resolution_status": "not-reported",
        "pipeline_version": "1.4.0",
        "prompt_version": "paper-evidence-local-v17",
        "source_version_id": "ab-bind-antibody-binding-mutational-database-for-compu-2015-11-06",
        "verifier_model_requested": "gpt-5.6-sol",
        "verifier_model_resolved": null
      },
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": null,
          "canonical_url": "https://doi.org/10.1002/pro.2829",
          "content_sha256": "8f17a3c32d5a5afeb66bfe2cf2f1311252145a5f0e284aebd35dc1f5e45b6355",
          "content_type": "application/pdf",
          "doi": "10.1002/pro.2829",
          "id": "ab-bind-antibody-binding-mutational-database-for-compu-2015-11-06",
          "label": "Reviewed source",
          "publication_date": "2015-11-06",
          "retrieved_at": "2026-08-14T05:40:07+00:00",
          "source_access": "submitted-pdf",
          "status": "version-of-record"
        }
      ],
      "status": "published",
      "title": "AB-Bind: Antibody binding mutational database for computational affinity predictions",
      "verification": {
        "last_verified": "2026-08-14",
        "notes": "AI-assisted double-pass extraction; production inclusion still requires the owner SHA-comment gate.",
        "status": "verified"
      },
      "work_type": "paper"
    },
    {
      "arxiv": "2506.04235",
      "authors": [
        "Xinyan Zhao",
        "Yi-Ching Tang",
        "Akshita Singh",
        "Victor J Cantu",
        "KwanHo An",
        "Junseok Lee",
        "Adam E Stogsdill",
        "Ibraheem M Hamdi",
        "Ashwin Kumar Ramesh",
        "Zhiqiang An",
        "Xiaoqian Jiang",
        "Yejin Kim"
      ],
      "benchmark_ids": [
        "abbibench"
      ],
      "benchmark_use_ids": [
        "abbibench-a-benchmark-for-antibody-binding-affinity-ma-abbibench-1-use",
        "abbibench-a-benchmark-for-antibody-binding-affinity-ma-abbibench-2-use"
      ],
      "canonical_url": "https://arxiv.org/abs/2506.04235",
      "current_version_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma-arxiv-v2",
      "doi": null,
      "entity_type": "work",
      "evaluation_run_ids": [],
      "id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma",
      "organizations": [
        "McWilliams School of Biomedical Informatics, UTHealth Houston",
        "Department of Industrial and Systems Engineering, Korea Advanced Institute of Science and Technology",
        "Texas Therapeutics Institute, Brown Foundation Institute of Molecular Medicine, UTHealth Houston"
      ],
      "publication_date": "2025-05-23",
      "review_provenance": {
        "codex_cli_version": "codex-cli 0.148.0-alpha.9",
        "execution_surface": "local-codex-cli",
        "extractor_model_requested": "gpt-5.6-sol",
        "extractor_model_resolved": null,
        "generated_at": "2026-08-20T06:31:06+00:00",
        "local_run_id": "7f0394d0-797e-402d-b2f2-07d324b56312",
        "method": "local-codex-double-pass",
        "model_resolution_status": "not-reported",
        "pipeline_version": "1.4.0",
        "prompt_version": "paper-evidence-local-v17",
        "source_version_id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma-arxiv-v2",
        "verifier_model_requested": "gpt-5.6-sol",
        "verifier_model_resolved": null
      },
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": "2506.04235",
          "canonical_url": "https://arxiv.org/abs/2506.04235",
          "content_sha256": "7009bb93e8eba70952d62635852ecb3b514aa760166771f5330134fd20712a2f",
          "content_type": "application/pdf",
          "doi": null,
          "id": "abbibench-a-benchmark-for-antibody-binding-affinity-ma-arxiv-v2",
          "label": "arXiv v2",
          "publication_date": "2025-05-23",
          "retrieved_at": "2026-08-20T06:31:06+00:00",
          "source_access": "submitted-pdf",
          "status": "current"
        }
      ],
      "status": "preprint",
      "title": "AbBiBench: A Benchmark for Antibody Binding Affinity Maturation and Design",
      "verification": {
        "last_verified": "2026-08-20",
        "notes": "AI-assisted double-pass extraction; production inclusion still requires the owner SHA-comment gate.",
        "status": "verified"
      },
      "work_type": "preprint"
    },
    {
      "arxiv": null,
      "authors": [
        "Lu Hong",
        "Tanja Kortemme"
      ],
      "benchmark_ids": [
        "cam-benchmark",
        "papd-benchmark",
        "rfah-benchmark"
      ],
      "benchmark_use_ids": [
        "an-integrative-approach-to-protein-sequence-design-thr-cam-benchmark-5-use",
        "an-integrative-approach-to-protein-sequence-design-thr-cam-benchmark-6-use",
        "an-integrative-approach-to-protein-sequence-design-thr-papd-benchmark-3-use",
        "an-integrative-approach-to-protein-sequence-design-thr-papd-benchmark-4-use",
        "an-integrative-approach-to-protein-sequence-design-thr-rfah-benchmark-1-use",
        "an-integrative-approach-to-protein-sequence-design-thr-rfah-benchmark-2-use"
      ],
      "canonical_url": "https://doi.org/10.1371/journal.pcbi.1011953",
      "current_version_id": "an-integrative-approach-to-protein-sequence-design-thr-pmc-version-1",
      "doi": "10.1371/journal.pcbi.1011953",
      "entity_type": "work",
      "evaluation_run_ids": [],
      "id": "an-integrative-approach-to-protein-sequence-design-thr",
      "organizations": [
        "University of California, San Francisco",
        "Quantitative Biosciences Institute",
        "Chan Zuckerberg Biohub"
      ],
      "publication_date": "2024-07-11",
      "review_provenance": {
        "codex_cli_version": "codex-cli 0.146.0-alpha.9.2",
        "execution_surface": "local-codex-cli",
        "extractor_model_requested": "gpt-5.6-sol",
        "extractor_model_resolved": null,
        "generated_at": "2026-07-31T05:54:17+00:00",
        "local_run_id": "1d7378b3-42fd-42ba-a2c3-9d90f8335349",
        "method": "local-codex-double-pass",
        "model_resolution_status": "not-reported",
        "pipeline_version": "1.4.0",
        "prompt_version": "paper-evidence-local-v15",
        "source_version_id": "an-integrative-approach-to-protein-sequence-design-thr-pmc-version-1",
        "verifier_model_requested": "gpt-5.6-sol",
        "verifier_model_resolved": null
      },
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": null,
          "canonical_url": "https://doi.org/10.1371/journal.pcbi.1011953",
          "content_sha256": "8a6d1008e88beef51751dafeafed80b607afcacb081a94f18de3f8d3d9d39814",
          "content_type": "application/xml",
          "doi": "10.1371/journal.pcbi.1011953",
          "id": "an-integrative-approach-to-protein-sequence-design-thr-pmc-version-1",
          "label": "PMC version 1",
          "publication_date": "2024-07-11",
          "retrieved_at": "2026-07-31T05:54:17+00:00",
          "source_access": "open-url",
          "status": "version-of-record"
        }
      ],
      "status": "published",
      "title": "An integrative approach to protein sequence design through multiobjective optimization",
      "verification": {
        "last_verified": "2026-07-31",
        "notes": "AI-assisted double-pass extraction; production inclusion still requires the owner SHA-comment gate.",
        "status": "verified"
      },
      "work_type": "paper"
    },
    {
      "arxiv": null,
      "authors": [
        "Anthropic"
      ],
      "benchmark_ids": [
        "anthropic-computational-biology",
        "anthropic-key-life-sciences-evals",
        "anthropic-protein-understanding",
        "anthropic-scientific-figure-interpretation",
        "spatialbench"
      ],
      "benchmark_use_ids": [
        "anthropic-computational-biology-evaluation",
        "anthropic-key-life-sciences-creation",
        "anthropic-protein-understanding-evaluation",
        "anthropic-scientific-figure-evaluation",
        "anthropic-spatialbench-external-summary"
      ],
      "canonical_url": "https://www.anthropic.com/news/healthcare-life-sciences",
      "current_version_id": "anthropic-healthcare-life-sciences-2026-01-11",
      "doi": null,
      "entity_type": "work",
      "evaluation_run_ids": [
        "anthropic-computational-biology-delta",
        "anthropic-protein-understanding-delta",
        "anthropic-scientific-figure-delta"
      ],
      "id": "anthropic-healthcare-life-sciences",
      "organizations": [
        "Anthropic"
      ],
      "publication_date": "2026-01-11",
      "source_class": "official_model_provider",
      "source_versions": [
        {
          "arxiv": null,
          "canonical_url": "https://www.anthropic.com/news/healthcare-life-sciences",
          "doi": null,
          "id": "anthropic-healthcare-life-sciences-2026-01-11",
          "label": "Canonical source",
          "publication_date": "2026-01-11",
          "status": "current"
        }
      ],
      "status": "published",
      "title": "Advancing Claude in healthcare and the life sciences",
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Official model-provider page; its SpatialBench panel explicitly attributes results to LatchBio, while its key-life-sciences panel reports Anthropic internal evaluations.",
        "status": "verified"
      },
      "work_type": "official-release"
    },
    {
      "arxiv": null,
      "authors": [
        "Anthropic"
      ],
      "benchmark_ids": [
        "bixbench",
        "lab-bench-protocolqa"
      ],
      "benchmark_use_ids": [
        "anthropic-life-sciences-bixbench"
      ],
      "canonical_url": "https://www.anthropic.com/news/claude-for-life-sciences",
      "current_version_id": "anthropic-life-sciences-2025-10-20",
      "doi": null,
      "entity_type": "work",
      "evaluation_run_ids": [
        "lab-bench-protocolqa-anthropic"
      ],
      "id": "anthropic-life-sciences",
      "organizations": [
        "Anthropic"
      ],
      "publication_date": "2025-10-20",
      "source_class": "official_model_provider",
      "source_versions": [
        {
          "arxiv": null,
          "canonical_url": "https://www.anthropic.com/news/claude-for-life-sciences",
          "doi": null,
          "id": "anthropic-life-sciences-2025-10-20",
          "label": "Canonical source",
          "publication_date": "2025-10-20",
          "status": "current"
        }
      ],
      "status": "published",
      "title": "Claude for Life Sciences",
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Official model-provider release with ProtocolQA details and a partial BixBench comparison claim.",
        "status": "verified"
      },
      "work_type": "official-release"
    },
    {
      "arxiv": null,
      "authors": [
        "Anthropic"
      ],
      "benchmark_ids": [
        "lab-bench-cloning-scenarios",
        "lab-bench-figqa",
        "lab-bench-protocolqa",
        "lab-bench-seqqa"
      ],
      "benchmark_use_ids": [],
      "canonical_url": "https://assets.anthropic.com/m/12f214efcc2f457a/original/Claude-Sonnet-4-5-System-Card.pdf",
      "current_version_id": "anthropic-sonnet-4-5-system-card-2025-09-29",
      "doi": null,
      "entity_type": "work",
      "evaluation_run_ids": [
        "lab-bench-cloning-scenarios-anthropic-sonnet45-system-card",
        "lab-bench-figqa-anthropic-sonnet45-system-card",
        "lab-bench-protocolqa-anthropic-sonnet45-system-card",
        "lab-bench-seqqa-anthropic-sonnet45-system-card"
      ],
      "id": "anthropic-sonnet-4-5-system-card",
      "organizations": [
        "Anthropic"
      ],
      "publication_date": "2025-09-29",
      "source_class": "official_model_provider",
      "source_versions": [
        {
          "arxiv": null,
          "canonical_url": "https://assets.anthropic.com/m/12f214efcc2f457a/original/Claude-Sonnet-4-5-System-Card.pdf",
          "doi": null,
          "id": "anthropic-sonnet-4-5-system-card-2025-09-29",
          "label": "Canonical source",
          "publication_date": "2025-09-29",
          "status": "current"
        }
      ],
      "status": "published",
      "title": "Claude Sonnet 4.5 System Card",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Official PDF, SHA256 d5d1ce28fea101ae3b201a5d4da100090a5d6b73246f7da820cafd7dbb11100e; LAB-Bench details and figure are on printed pages 132–133.",
        "status": "verified"
      },
      "work_type": "system-card"
    },
    {
      "arxiv": null,
      "authors": [
        "Anthropic"
      ],
      "benchmark_ids": [
        "lab-bench-figqa"
      ],
      "benchmark_use_ids": [],
      "canonical_url": "https://www.anthropic.com/claude-sonnet-4-6-system-card",
      "current_version_id": "anthropic-sonnet-4-6-system-card-2026-02-17",
      "doi": null,
      "entity_type": "work",
      "evaluation_run_ids": [
        "lab-bench-figqa-crop-tool",
        "lab-bench-figqa-no-tools"
      ],
      "id": "anthropic-sonnet-4-6-system-card",
      "organizations": [
        "Anthropic"
      ],
      "publication_date": "2026-02-17",
      "source_class": "official_model_provider",
      "source_versions": [
        {
          "arxiv": null,
          "canonical_url": "https://www.anthropic.com/claude-sonnet-4-6-system-card",
          "doi": null,
          "id": "anthropic-sonnet-4-6-system-card-2026-02-17",
          "label": "Canonical source",
          "publication_date": "2026-02-17",
          "status": "current"
        }
      ],
      "status": "published",
      "title": "Claude Sonnet 4.6 System Card",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Official model-provider system card with FigQA tool/no-tool results.",
        "status": "verified"
      },
      "work_type": "system-card"
    },
    {
      "arxiv": "2012.04035",
      "authors": [
        "Raphael J. L. Townshend",
        "Martin Vögele",
        "Patricia Suriana",
        "Alexander Derry",
        "Alexander Powers",
        "Yianni Laloudakis",
        "Sidhika Balachandar",
        "Bowen Jing",
        "Brandon Anderson",
        "Stephan Eismann",
        "Risi Kondor",
        "Russ B. Altman",
        "Ron O. Dror"
      ],
      "benchmark_ids": [
        "atom3d"
      ],
      "benchmark_use_ids": [],
      "canonical_url": "https://datasets-benchmarks-proceedings.neurips.cc/paper/2021/hash/c45147dee729311ef5b5c3003946c48f-Abstract-round1.html",
      "current_version_id": "atom3d-paper-2021-12-07",
      "doi": null,
      "entity_type": "work",
      "evaluation_run_ids": [
        "atom3d-creator-full"
      ],
      "id": "atom3d-paper",
      "organizations": [
        "Stanford University"
      ],
      "publication_date": "2021-12-07",
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": "2012.04035",
          "canonical_url": "https://datasets-benchmarks-proceedings.neurips.cc/paper/2021/hash/c45147dee729311ef5b5c3003946c48f-Abstract-round1.html",
          "doi": null,
          "id": "atom3d-paper-2021-12-07",
          "label": "Canonical source",
          "publication_date": "2021-12-07",
          "status": "current"
        }
      ],
      "status": "published",
      "title": "ATOM3D: Tasks On Molecules in Three Dimensions",
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "NeurIPS 2021 Datasets and Benchmarks creator paper and official project resources verified.",
        "status": "verified"
      },
      "work_type": "paper"
    },
    {
      "arxiv": "2406.10391",
      "authors": [
        "Yuchen Ren",
        "Zhiyuan Chen",
        "Lifeng Qiao",
        "Hongtai Jing",
        "Yuchen Cai",
        "Sheng Xu",
        "Peng Ye",
        "Xinzhu Ma",
        "Siqi Sun",
        "Hongliang Yan",
        "Dong Yuan",
        "Wanli Ouyang",
        "Xihui Liu"
      ],
      "benchmark_ids": [
        "beacon-rna"
      ],
      "benchmark_use_ids": [],
      "canonical_url": "https://proceedings.neurips.cc/paper_files/paper/2024/hash/a8ea503d91320fcfe12cba61c8a6d285-Abstract-Datasets_and_Benchmarks_Track.html",
      "current_version_id": "beacon-paper-2024-12-10",
      "doi": null,
      "entity_type": "work",
      "evaluation_run_ids": [
        "beacon-creator-full"
      ],
      "id": "beacon-paper",
      "organizations": [
        "Shanghai Artificial Intelligence Laboratory",
        "University of Sydney",
        "University of Hong Kong",
        "Shanghai Jiao Tong University",
        "Fudan University",
        "Chinese University of Hong Kong"
      ],
      "publication_date": "2024-12-10",
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": "2406.10391",
          "canonical_url": "https://proceedings.neurips.cc/paper_files/paper/2024/hash/a8ea503d91320fcfe12cba61c8a6d285-Abstract-Datasets_and_Benchmarks_Track.html",
          "doi": null,
          "id": "beacon-paper-2024-12-10",
          "label": "Canonical source",
          "publication_date": "2024-12-10",
          "status": "current"
        }
      ],
      "status": "published",
      "title": "BEACON: Benchmark for Comprehensive RNA Tasks and Language Models",
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Final NeurIPS 2024 Datasets and Benchmarks paper and official repository verified.",
        "status": "verified"
      },
      "work_type": "paper"
    },
    {
      "arxiv": "2412.19191",
      "authors": [
        "Haonan He",
        "Yuchen Ren",
        "Yining Tang",
        "Ziyang Xu",
        "Junxian Li",
        "Minghao Yang",
        "Di Zhang",
        "Yuan Dong",
        "Tao Chen",
        "Shufei Zhang",
        "Yuqiang Li",
        "Nanqing Dong",
        "Wanli Ouyang",
        "Dongzhan Zhou",
        "Peng Ye"
      ],
      "benchmark_ids": [
        "bioinstruction-aan",
        "bioinstruction-apa",
        "bioinstruction-cpd",
        "bioinstruction-crispr-on-target",
        "bioinstruction-ea",
        "bioinstruction-ec",
        "bioinstruction-emp",
        "bioinstruction-epi",
        "bioinstruction-fluorescence",
        "bioinstruction-modification",
        "bioinstruction-mrl",
        "bioinstruction-ncrna",
        "bioinstruction-pd300",
        "bioinstruction-prs",
        "bioinstruction-rpi",
        "bioinstruction-sirna",
        "bioinstruction-solubility",
        "bioinstruction-stability",
        "bioinstruction-tb-human",
        "bioinstruction-tb-mouse",
        "bioinstruction-thermostability"
      ],
      "benchmark_use_ids": [],
      "canonical_url": "https://aclanthology.org/2025.findings-emnlp.978/",
      "current_version_id": "bioinstructions-paper-2025-11-04",
      "doi": "10.18653/v1/2025.findings-emnlp.978",
      "entity_type": "work",
      "evaluation_run_ids": [
        "bioinstruction-aan-closed-baselines",
        "bioinstruction-aan-creator-systems",
        "bioinstruction-aan-open-baselines",
        "bioinstruction-apa-closed-baselines",
        "bioinstruction-apa-creator-systems",
        "bioinstruction-apa-open-baselines",
        "bioinstruction-cpd-closed-baselines",
        "bioinstruction-cpd-creator-systems",
        "bioinstruction-cpd-open-baselines",
        "bioinstruction-crispr-on-target-closed-baselines",
        "bioinstruction-crispr-on-target-creator-systems",
        "bioinstruction-crispr-on-target-open-baselines",
        "bioinstruction-ea-closed-baselines",
        "bioinstruction-ea-creator-systems",
        "bioinstruction-ea-open-baselines",
        "bioinstruction-ec-closed-baselines",
        "bioinstruction-ec-creator-systems",
        "bioinstruction-ec-open-baselines",
        "bioinstruction-emp-closed-baselines",
        "bioinstruction-emp-creator-systems",
        "bioinstruction-emp-open-baselines",
        "bioinstruction-epi-closed-baselines",
        "bioinstruction-epi-creator-systems",
        "bioinstruction-epi-open-baselines",
        "bioinstruction-fluorescence-closed-baselines",
        "bioinstruction-fluorescence-creator-systems",
        "bioinstruction-fluorescence-open-baselines",
        "bioinstruction-modification-closed-baselines",
        "bioinstruction-modification-creator-systems",
        "bioinstruction-modification-open-baselines",
        "bioinstruction-mrl-closed-baselines",
        "bioinstruction-mrl-creator-systems",
        "bioinstruction-mrl-open-baselines",
        "bioinstruction-ncrna-closed-baselines",
        "bioinstruction-ncrna-creator-systems",
        "bioinstruction-ncrna-open-baselines",
        "bioinstruction-pd300-closed-baselines",
        "bioinstruction-pd300-creator-systems",
        "bioinstruction-pd300-open-baselines",
        "bioinstruction-prs-closed-baselines",
        "bioinstruction-prs-creator-systems",
        "bioinstruction-prs-open-baselines",
        "bioinstruction-rpi-closed-baselines",
        "bioinstruction-rpi-creator-systems",
        "bioinstruction-rpi-open-baselines",
        "bioinstruction-sirna-closed-baselines",
        "bioinstruction-sirna-creator-systems",
        "bioinstruction-sirna-open-baselines",
        "bioinstruction-solubility-closed-baselines",
        "bioinstruction-solubility-creator-systems",
        "bioinstruction-solubility-open-baselines",
        "bioinstruction-stability-closed-baselines",
        "bioinstruction-stability-creator-systems",
        "bioinstruction-stability-open-baselines",
        "bioinstruction-tb-human-closed-baselines",
        "bioinstruction-tb-human-creator-systems",
        "bioinstruction-tb-human-open-baselines",
        "bioinstruction-tb-mouse-closed-baselines",
        "bioinstruction-tb-mouse-creator-systems",
        "bioinstruction-tb-mouse-open-baselines",
        "bioinstruction-thermostability-closed-baselines",
        "bioinstruction-thermostability-creator-systems",
        "bioinstruction-thermostability-open-baselines"
      ],
      "id": "bioinstructions-paper",
      "organizations": [
        "Shanghai Artificial Intelligence Laboratory",
        "University of Science and Technology of China",
        "University of Sydney",
        "University of Toronto",
        "Chinese University of Hong Kong",
        "Shanghai Jiao Tong University",
        "Fudan University",
        "Shanghai Innovation Institute"
      ],
      "publication_date": "2025-11-04",
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": "2412.19191",
          "canonical_url": "https://aclanthology.org/2025.findings-emnlp.978/",
          "doi": "10.18653/v1/2025.findings-emnlp.978",
          "id": "bioinstructions-paper-2025-11-04",
          "label": "Canonical source",
          "publication_date": "2025-11-04",
          "status": "current"
        }
      ],
      "status": "published",
      "title": "Biology-Instructions: A Dataset and Benchmark for Multi-Omics Sequence Understanding Capability of Large Language Models",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Final ACL Anthology record and paper footer report EMNLP 2025, November 4-9; the first conference date is used as the publication date.",
        "status": "verified"
      },
      "work_type": "paper"
    },
    {
      "arxiv": null,
      "authors": [
        "Brianna"
      ],
      "benchmark_ids": [
        "biomysterybench"
      ],
      "benchmark_use_ids": [],
      "canonical_url": "https://www.anthropic.com/research/Evaluating-Claude-For-Bioinformatics-With-BioMysteryBench",
      "current_version_id": "biomysterybench-official-2026-04-29",
      "doi": null,
      "entity_type": "work",
      "evaluation_run_ids": [
        "biomysterybench-official-run",
        "biomysterybench-v8-human-difficult",
        "biomysterybench-v8-human-solvable"
      ],
      "id": "biomysterybench-official",
      "organizations": [
        "Anthropic"
      ],
      "publication_date": "2026-04-29",
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": null,
          "canonical_url": "https://www.anthropic.com/research/Evaluating-Claude-For-Bioinformatics-With-BioMysteryBench",
          "doi": null,
          "id": "biomysterybench-official-2026-04-29",
          "label": "Canonical source",
          "publication_date": "2026-04-29",
          "status": "current"
        }
      ],
      "status": "living",
      "title": "Evaluating Claude's bioinformatics research capabilities with BioMysteryBench",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Official Anthropic creator report. The page identifies Brianna, a Discovery Team researcher, as the writer; no surname or formal author list is supplied.",
        "status": "verified"
      },
      "work_type": "official-release"
    },
    {
      "arxiv": "2607.19262",
      "authors": [
        "Harmon Bhasin",
        "Kevin Flyangolts",
        "Dianzhuo Wang",
        "Evan Seeyave",
        "Arjun Banerjee",
        "Amanda Darling",
        "Joshua Stallings",
        "David Stern",
        "Shawn Higdon",
        "Claire Duvallet",
        "Bryan Tegomoh",
        "Kenny Workman"
      ],
      "benchmark_ids": [
        "biosecbench-surveillance"
      ],
      "benchmark_use_ids": [
        "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-1-use",
        "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-2-use",
        "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-3-use",
        "biosecbench-surveillance-a-verifiable-benchmark-for-ai-biosecbench-surveillance-4-use"
      ],
      "canonical_url": "https://arxiv.org/abs/2607.19262",
      "current_version_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-arxiv-v1",
      "doi": null,
      "entity_type": "work",
      "evaluation_run_ids": [],
      "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai",
      "organizations": [
        "LatchBio",
        "Aclid"
      ],
      "publication_date": "2026-07-21",
      "review_provenance": {
        "codex_cli_version": "codex-cli 0.146.0-alpha.3.1",
        "execution_surface": "local-codex-cli",
        "extractor_model_requested": "gpt-5.6-sol",
        "extractor_model_resolved": null,
        "generated_at": "2026-07-30T22:18:35+00:00",
        "local_run_id": "b1b64922-a00e-4a61-a55c-3b7b21b4d465",
        "method": "local-codex-double-pass",
        "model_resolution_status": "not-reported",
        "pipeline_version": "1.4.0",
        "prompt_version": "paper-evidence-local-v14",
        "source_version_id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-arxiv-v1",
        "verifier_model_requested": "gpt-5.6-sol",
        "verifier_model_resolved": null
      },
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": "2607.19262",
          "canonical_url": "https://arxiv.org/abs/2607.19262",
          "content_sha256": "0b4c701a9a04234b918d5e6b3392de3b3769ccd0941226f04ae0e39675447d7d",
          "content_type": "application/pdf",
          "doi": null,
          "id": "biosecbench-surveillance-a-verifiable-benchmark-for-ai-arxiv-v1",
          "label": "arXiv v1",
          "publication_date": "2026-07-21",
          "retrieved_at": "2026-07-30T22:18:35+00:00",
          "source_access": "submitted-pdf",
          "status": "current"
        }
      ],
      "status": "preprint",
      "title": "BioSecBench-Surveillance: A Verifiable Benchmark for AI Agents in Pathogen Genomic Surveillance",
      "verification": {
        "last_verified": "2026-07-30",
        "notes": "AI-assisted double-pass extraction; production inclusion still requires the owner SHA-comment gate.",
        "status": "verified"
      },
      "work_type": "preprint"
    },
    {
      "arxiv": null,
      "authors": [
        "LatchBio"
      ],
      "benchmark_ids": [
        "biosecbench-surveillance"
      ],
      "benchmark_use_ids": [
        "biosecbench-8d53fd8-claude-code-use",
        "biosecbench-8d53fd8-openai-codex-use",
        "biosecbench-8d53fd8-pi-use"
      ],
      "canonical_url": "https://github.com/latchbio/biosecbench-surveillance/tree/8d53fd8517cc74202eb18b618e8b39b4ffaf0c87",
      "current_version_id": "biosecbench-surveillance-repository-result-snapshot-8d53fd8",
      "doi": null,
      "entity_type": "work",
      "evaluation_run_ids": [
        "biosecbench-8d53fd8-claude-code",
        "biosecbench-8d53fd8-openai-codex",
        "biosecbench-8d53fd8-pi"
      ],
      "id": "biosecbench-surveillance-repository-result-snapshot",
      "organizations": [
        "LatchBio"
      ],
      "publication_date": "2026-07-09",
      "review_provenance": {
        "codex_cli_version": "codex-cli 0.147.0-alpha.6.5",
        "execution_surface": "local-codex-cli",
        "extractor_model_requested": "gpt-5.6-sol",
        "extractor_model_resolved": null,
        "generated_at": "2026-08-11T09:29:54Z",
        "local_run_id": "71a96e8c-edfd-4bb0-a0f7-4197d05b5342",
        "method": "local-codex-double-pass",
        "model_resolution_status": "not-reported",
        "pipeline_version": "1.4.0",
        "prompt_version": "paper-evidence-local-v17",
        "source_version_id": "biosecbench-surveillance-repository-result-snapshot-8d53fd8",
        "verifier_model_requested": "gpt-5.6-sol",
        "verifier_model_resolved": null
      },
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": null,
          "canonical_url": "https://github.com/latchbio/biosecbench-surveillance/tree/8d53fd8517cc74202eb18b618e8b39b4ffaf0c87",
          "content_sha256": "27ef42fc30608b4f73085cbcc355f6e631e4cc0d8bdf1195d9c5558a174fa8e6",
          "content_type": "text/plain",
          "doi": null,
          "id": "biosecbench-surveillance-repository-result-snapshot-8d53fd8",
          "label": "Repository snapshot 8d53fd8",
          "publication_date": "2026-07-09",
          "retrieved_at": "2026-08-11T09:29:54Z",
          "source_access": "open-url",
          "status": "current"
        }
      ],
      "status": "published",
      "title": "BioSecBench-Surveillance repository result snapshot",
      "verification": {
        "last_verified": "2026-08-11",
        "notes": "Commit-pinned official result snapshot; all published claims passed independent local double-pass verification.",
        "status": "verified"
      },
      "work_type": "official-release"
    },
    {
      "arxiv": "2503.00096",
      "authors": [
        "Ludovico Mitchener",
        "Jon M. Laurent",
        "Benjamin Tenmann",
        "Siddharth Narayanan",
        "Geemi P. Wellawatte",
        "Andrew White",
        "Lorenzo Sani",
        "Samuel G. Rodriques"
      ],
      "benchmark_ids": [
        "bixbench"
      ],
      "benchmark_use_ids": [],
      "canonical_url": "https://arxiv.org/abs/2503.00096",
      "current_version_id": "bixbench-paper-2025-02-28",
      "doi": null,
      "entity_type": "work",
      "evaluation_run_ids": [
        "bixbench-creator-paper",
        "bixbench-paper-mcq-no-images",
        "bixbench-paper-mcq-no-refusal",
        "bixbench-paper-mcq-refusal"
      ],
      "id": "bixbench-paper",
      "organizations": [
        "FutureHouse",
        "ScienceMachine"
      ],
      "publication_date": "2025-02-28",
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": "2503.00096",
          "canonical_url": "https://arxiv.org/abs/2503.00096",
          "doi": null,
          "id": "bixbench-paper-2025-02-28",
          "label": "Canonical source",
          "publication_date": "2025-02-28",
          "status": "current"
        }
      ],
      "status": "preprint",
      "title": "BixBench: a Comprehensive Benchmark for LLM-based Agents in Computational Biology",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Title, eight-author list, initial submission date, and arXiv identifier verified against the creator-hosted PDF and arXiv record; evaluations refer to the v1.0 paper snapshot.",
        "status": "verified"
      },
      "work_type": "preprint"
    },
    {
      "arxiv": null,
      "authors": [
        "FutureHouse BixBench maintainers"
      ],
      "benchmark_ids": [
        "bixbench"
      ],
      "benchmark_use_ids": [],
      "canonical_url": "https://github.com/Future-House/BixBench/tree/28909d842bc492ecd99bab303279afb29e3cb353",
      "current_version_id": "bixbench-v1-5-release-2025-09-26",
      "doi": null,
      "entity_type": "work",
      "evaluation_run_ids": [
        "bixbench-v1-5-agentic-mcq-no-refusal-images",
        "bixbench-v1-5-agentic-mcq-refusal-images",
        "bixbench-v1-5-agentic-mcq-refusal-no-images",
        "bixbench-v1-5-agentic-open-images",
        "bixbench-v1-5-zero-shot-mcq-no-refusal",
        "bixbench-v1-5-zero-shot-mcq-refusal",
        "bixbench-v1-5-zero-shot-open"
      ],
      "id": "bixbench-v1-5-release",
      "organizations": [
        "FutureHouse",
        "ScienceMachine"
      ],
      "publication_date": "2025-09-26",
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": null,
          "canonical_url": "https://github.com/Future-House/BixBench/tree/28909d842bc492ecd99bab303279afb29e3cb353",
          "doi": null,
          "id": "bixbench-v1-5-release-2025-09-26",
          "label": "Canonical source",
          "publication_date": "2025-09-26",
          "status": "current"
        }
      ],
      "status": "published",
      "title": "BixBench v1.5 dataset and evaluation release",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Creator repository commit adds v1.5 evaluation results; the related v1.5 dataset-card update is dated 2025-09-23 and its tag commit is dated 2025-09-29.",
        "status": "verified"
      },
      "work_type": "official-release"
    },
    {
      "arxiv": "2408.09667",
      "authors": [
        "Ken Gu",
        "Ruoxi Shang",
        "Ruien Jiang",
        "Keying Kuang",
        "Richard-John Lin",
        "Donghe Lyu",
        "Yue Mao",
        "Youran Pan",
        "Teng Wu",
        "Jiaqian Yu",
        "Yikun Zhang",
        "Tianmai M. Zhang",
        "Lanyi Zhu",
        "Mike A. Merrill",
        "Jeffrey Heer",
        "Tim Althoff"
      ],
      "benchmark_ids": [
        "blade-analysis-generation",
        "blade-mcq"
      ],
      "benchmark_use_ids": [],
      "canonical_url": "https://aclanthology.org/2024.findings-emnlp.815/",
      "current_version_id": "blade-paper-2024-11-16",
      "doi": "10.18653/v1/2024.findings-emnlp.815",
      "entity_type": "work",
      "evaluation_run_ids": [
        "blade-creator-decision-mcq",
        "blade-creator-paper",
        "blade-creator-react"
      ],
      "id": "blade-paper",
      "organizations": [
        "University of Washington",
        "UC Berkeley",
        "New York University",
        "Stanford University",
        "University of British Columbia",
        "Microsoft",
        "George Washington University"
      ],
      "publication_date": "2024-11-16",
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": "2408.09667",
          "canonical_url": "https://aclanthology.org/2024.findings-emnlp.815/",
          "doi": "10.18653/v1/2024.findings-emnlp.815",
          "id": "blade-paper-2024-11-16",
          "label": "Canonical source",
          "publication_date": "2024-11-16",
          "status": "current"
        }
      ],
      "status": "published",
      "title": "BLADE: Benchmarking Language Model Agents for Data-Driven Science",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Title, sixteen authors, seven represented organizations, DOI, arXiv identifier, and Findings of EMNLP 2024 publication were verified; current count and protocol evidence uses the linked creator manuscript v3 where stated.",
        "status": "verified"
      },
      "work_type": "paper"
    },
    {
      "arxiv": null,
      "authors": [
        "Jürgen Haas",
        "Alessandro Barbato",
        "Dario Behringer",
        "Gabriel Studer",
        "Steven Roth",
        "Marco Bertoni",
        "Kamila Mostaguir",
        "Rafal Gumienny",
        "Torsten Schwede"
      ],
      "benchmark_ids": [],
      "benchmark_use_ids": [],
      "canonical_url": "https://doi.org/10.1002/prot.25431",
      "current_version_id": "cameo-foundation-paper-2017-12-17",
      "doi": "10.1002/prot.25431",
      "entity_type": "work",
      "evaluation_run_ids": [],
      "id": "cameo-foundation-paper",
      "organizations": [
        "SIB Swiss Institute of Bioinformatics",
        "Biozentrum University of Basel"
      ],
      "publication_date": "2017-12-17",
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": null,
          "canonical_url": "https://doi.org/10.1002/prot.25431",
          "doi": "10.1002/prot.25431",
          "id": "cameo-foundation-paper-2017-12-17",
          "label": "Canonical source",
          "publication_date": "2017-12-17",
          "status": "current"
        }
      ],
      "status": "published",
      "title": "Continuous Automated Model EvaluatiOn (CAMEO) Complementing the Critical Assessment of Structure Prediction in CASP12",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Creator paper for the original weekly continuous platform; final journal issue is Proteins 86(S1):387-398 (2018).",
        "status": "verified"
      },
      "work_type": "paper"
    },
    {
      "arxiv": null,
      "authors": [
        "Xavier Robin",
        "Peter Škrinjar",
        "Andrew M. Waterhouse",
        "Gabriel Studer",
        "Gerardo Tauriello",
        "Janani Durairaj",
        "Torsten Schwede"
      ],
      "benchmark_ids": [
        "cameo"
      ],
      "benchmark_use_ids": [],
      "canonical_url": "https://doi.org/10.1002/prot.70060",
      "current_version_id": "cameo-paper-2025-09-28",
      "doi": "10.1002/prot.70060",
      "entity_type": "work",
      "evaluation_run_ids": [
        "cameo-2024-antibody-three-server-common",
        "cameo-2024-ligand-baseline-common",
        "cameo-2024-ppi-three-server-common"
      ],
      "id": "cameo-paper",
      "organizations": [
        "SIB Swiss Institute of Bioinformatics",
        "Biozentrum University of Basel"
      ],
      "publication_date": "2025-09-28",
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": null,
          "canonical_url": "https://doi.org/10.1002/prot.70060",
          "doi": "10.1002/prot.70060",
          "id": "cameo-paper-2025-09-28",
          "label": "Canonical source",
          "publication_date": "2025-09-28",
          "status": "current"
        }
      ],
      "status": "published",
      "title": "Beyond Single Chains: Benchmarking Macromolecular Complex Prediction Methods With the Continuous Automated Model EvaluatiOn (CAMEO)",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Final creator paper; first published online 2025-09-28 and printed in Proteins 94(1):403-413.",
        "status": "verified"
      },
      "work_type": "paper"
    },
    {
      "arxiv": null,
      "authors": [
        "CASP organizing committee",
        "Prediction Center"
      ],
      "benchmark_ids": [],
      "benchmark_use_ids": [],
      "canonical_url": "https://predictioncenter.org/casp16/",
      "current_version_id": "casp-official-2024-05-01",
      "doi": null,
      "entity_type": "work",
      "evaluation_run_ids": [],
      "id": "casp-official",
      "organizations": [
        "Prediction Center",
        "CASP organizing committee"
      ],
      "publication_date": "2024-05-01",
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": null,
          "canonical_url": "https://predictioncenter.org/casp16/",
          "doi": null,
          "id": "casp-official-2024-05-01",
          "label": "Canonical source",
          "publication_date": "2024-05-01",
          "status": "current"
        }
      ],
      "status": "archived",
      "title": "CASP16 official competition and results archive",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Completed official 2024 round portal with targets, protocol, predictions, result tables, rankings, abstracts, presentations, and archive links.",
        "status": "verified"
      },
      "work_type": "competition-site"
    },
    {
      "arxiv": null,
      "authors": [
        "Michael K. Gilson",
        "Jerome Eberhardt",
        "Peter Škrinjar",
        "Janani Durairaj",
        "Xavier Robin",
        "Andriy Kryshtafovych"
      ],
      "benchmark_ids": [
        "casp-protein-ligands"
      ],
      "benchmark_use_ids": [],
      "canonical_url": "https://doi.org/10.1002/prot.70061",
      "current_version_id": "casp16-ligand-assessment-2025-10-04",
      "doi": "10.1002/prot.70061",
      "entity_type": "work",
      "evaluation_run_ids": [
        "casp16-ligand-affinity-stage1",
        "casp16-ligand-affinity-stage2",
        "casp16-ligand-pose-regular"
      ],
      "id": "casp16-ligand-assessment",
      "organizations": [
        "University of California San Diego",
        "University of Basel",
        "University of California Davis"
      ],
      "publication_date": "2025-10-04",
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": null,
          "canonical_url": "https://doi.org/10.1002/prot.70061",
          "doi": "10.1002/prot.70061",
          "id": "casp16-ligand-assessment-2025-10-04",
          "label": "Canonical source",
          "publication_date": "2025-10-04",
          "status": "current"
        }
      ],
      "status": "published",
      "title": "Assessment of Pharmaceutical Protein-Ligand Pose and Affinity Predictions in CASP16",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Final creator-assessor paper; reports 229 evaluated pose targets and 140 affinity targets across five protein systems.",
        "status": "verified"
      },
      "work_type": "paper"
    },
    {
      "arxiv": null,
      "authors": [
        "Rongqing Yuan",
        "Jing Zhang",
        "Andriy Kryshtafovych",
        "R. Dustin Schaeffer",
        "Jian Zhou",
        "Qian Cong",
        "Nick V. Grishin"
      ],
      "benchmark_ids": [
        "casp-protein-monomers"
      ],
      "benchmark_use_ids": [],
      "canonical_url": "https://doi.org/10.1002/prot.70031",
      "current_version_id": "casp16-monomer-assessment-2025-08-17",
      "doi": "10.1002/prot.70031",
      "entity_type": "work",
      "evaluation_run_ids": [
        "casp16-monomer-regular-official"
      ],
      "id": "casp16-monomer-assessment",
      "organizations": [
        "University of Texas Southwestern Medical Center",
        "University of California Davis"
      ],
      "publication_date": "2025-08-17",
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": null,
          "canonical_url": "https://doi.org/10.1002/prot.70031",
          "doi": "10.1002/prot.70031",
          "id": "casp16-monomer-assessment-2025-08-17",
          "label": "Canonical source",
          "publication_date": "2025-08-17",
          "status": "current"
        }
      ],
      "status": "published",
      "title": "CASP16 Protein Monomer Structure Prediction Assessment",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Final creator-assessor paper in the CASP16 special issue.",
        "status": "verified"
      },
      "work_type": "paper"
    },
    {
      "arxiv": null,
      "authors": [
        "Jing Zhang",
        "Rongqing Yuan",
        "Andriy Kryshtafovych",
        "Jimin Pei",
        "Rachael C. Kretsch",
        "R. Dustin Schaeffer",
        "Jian Zhou",
        "Rhiju Das",
        "Nick V. Grishin",
        "Qian Cong"
      ],
      "benchmark_ids": [
        "casp-protein-multimers"
      ],
      "benchmark_use_ids": [],
      "canonical_url": "https://doi.org/10.1002/prot.70068",
      "current_version_id": "casp16-multimer-assessment-2025-10-31",
      "doi": "10.1002/prot.70068",
      "entity_type": "work",
      "evaluation_run_ids": [
        "casp16-multimer-phase1-regular"
      ],
      "id": "casp16-multimer-assessment",
      "organizations": [
        "University of Texas Southwestern Medical Center",
        "University of California Davis",
        "Stanford University"
      ],
      "publication_date": "2025-10-31",
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": null,
          "canonical_url": "https://doi.org/10.1002/prot.70068",
          "doi": "10.1002/prot.70068",
          "id": "casp16-multimer-assessment-2025-10-31",
          "label": "Canonical source",
          "publication_date": "2025-10-31",
          "status": "current"
        }
      ],
      "status": "published",
      "title": "Assessment of Protein Complex Predictions in CASP16: Are We Making Progress?",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Final creator-assessor paper; distinguishes 40 unique Phase-1 targets from repeated Phase-0/2 releases.",
        "status": "verified"
      },
      "work_type": "paper"
    },
    {
      "arxiv": null,
      "authors": [
        "CASP organizing committee",
        "Prediction Center"
      ],
      "benchmark_ids": [],
      "benchmark_use_ids": [],
      "canonical_url": "https://predictioncenter.org/casp17/",
      "current_version_id": "casp17-official-2026-03-31",
      "doi": null,
      "entity_type": "work",
      "evaluation_run_ids": [],
      "id": "casp17-official",
      "organizations": [
        "Prediction Center",
        "CASP organizing committee"
      ],
      "publication_date": "2026-03-31",
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": null,
          "canonical_url": "https://predictioncenter.org/casp17/",
          "doi": null,
          "id": "casp17-official-2026-03-31",
          "label": "Canonical source",
          "publication_date": "2026-03-31",
          "status": "current"
        }
      ],
      "status": "living",
      "title": "CASP17 official competition",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Active 2026 round; targets and counts are rolling and results are not yet final.",
        "status": "verified"
      },
      "work_type": "competition-site"
    },
    {
      "arxiv": null,
      "authors": [
        "Surag Nair",
        "Laura Gunsalus",
        "Brian Orcutt-Jahns",
        "Jordan Rossen",
        "Avantika Lal",
        "Carlo De Donno",
        "Muhammed Hasan Çelik",
        "Kipper Fletez-Brant",
        "Xiaoman Xie",
        "Hector Corrada Bravo",
        "Gokcen Eraslan"
      ],
      "benchmark_ids": [
        "compbiobench"
      ],
      "benchmark_use_ids": [],
      "canonical_url": "https://www.biorxiv.org/content/10.64898/2026.04.06.716850v1",
      "current_version_id": "compbiobench-preprint-2026-04-09",
      "doi": "10.64898/2026.04.06.716850",
      "entity_type": "work",
      "evaluation_run_ids": [
        "compbiobench-codex-hardest",
        "compbiobench-creator-full",
        "compbiobench-haiku-full",
        "compbiobench-haiku-hardest",
        "compbiobench-nonagentic-baselines",
        "compbiobench-opus-full",
        "compbiobench-opus-hardest",
        "compbiobench-sonnet-full",
        "compbiobench-sonnet-hardest"
      ],
      "id": "compbiobench-preprint",
      "organizations": [
        "Genentech",
        "Roche"
      ],
      "publication_date": "2026-04-09",
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": null,
          "canonical_url": "https://www.biorxiv.org/content/10.64898/2026.04.06.716850v1",
          "doi": "10.64898/2026.04.06.716850",
          "id": "compbiobench-preprint-2026-04-09",
          "label": "Canonical source",
          "publication_date": "2026-04-09",
          "status": "current"
        }
      ],
      "status": "preprint",
      "title": "Agentic systems are adept at solving well-scoped, verifiable problems in computational biology",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Author list, posting date, DOI, v1 status, and official data/code links verified from bioRxiv v1.",
        "status": "verified"
      },
      "work_type": "preprint"
    },
    {
      "arxiv": null,
      "authors": [
        "Chit Tong Lio",
        "Tolga Düz",
        "Markus Hoffmann",
        "Lina-Liv Willruth",
        "Jan Baumbach",
        "Markus List",
        "Olga Tsoy"
      ],
      "benchmark_ids": [
        "comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu"
      ],
      "benchmark_use_ids": [
        "comprehensive-benchmark-of-differential-transcript-usa-comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-1-use",
        "comprehensive-benchmark-of-differential-transcript-usa-comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-2-use",
        "comprehensive-benchmark-of-differential-transcript-usa-comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-3-use",
        "comprehensive-benchmark-of-differential-transcript-usa-comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-4-use",
        "comprehensive-benchmark-of-differential-transcript-usa-comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-5-use",
        "comprehensive-benchmark-of-differential-transcript-usa-comprehensive-benchmark-of-differential-transcript-usage-analysis-for-bu-6-use"
      ],
      "canonical_url": "https://doi.org/10.1093/nargab/lqaf117",
      "current_version_id": "comprehensive-benchmark-of-differential-transcript-usa-final-published-version",
      "doi": "10.1093/nargab/lqaf117",
      "entity_type": "work",
      "evaluation_run_ids": [],
      "id": "comprehensive-benchmark-of-differential-transcript-usa",
      "organizations": [
        "Data Science in Systems Biology, Technical University of Munich",
        "Chair of Computational Systems Biology, University of Hamburg",
        "Institute for Advanced Study, Technical University of Munich",
        "National Institute of Diabetes, Digestive, and Kidney Diseases, National Institutes of Health",
        "Institute of Mathematics and Computer Science, University of Southern Denmark",
        "Munich Data Science Institute (MDSI), Technical University of Munich",
        "Department of Computer Science, Bioinformatics, Vrije Universiteit Amsterdam"
      ],
      "publication_date": "2025-09-11",
      "review_provenance": {
        "codex_cli_version": "codex-cli 0.147.0-alpha.1.2",
        "execution_surface": "local-codex-cli",
        "extractor_model_requested": "gpt-5.6-sol",
        "extractor_model_resolved": null,
        "generated_at": "2026-08-06T06:46:38+00:00",
        "local_run_id": "f8095a99-e3cb-4a8a-a850-34f241b5ee3a",
        "method": "local-codex-double-pass",
        "model_resolution_status": "not-reported",
        "pipeline_version": "1.4.0",
        "prompt_version": "paper-evidence-local-v17",
        "source_version_id": "comprehensive-benchmark-of-differential-transcript-usa-final-published-version",
        "verifier_model_requested": "gpt-5.6-sol",
        "verifier_model_resolved": null
      },
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": null,
          "canonical_url": "https://doi.org/10.1093/nargab/lqaf117",
          "content_sha256": "0b2e83ddecc33de7678f93e4f7d3eef21658a3ac996e9009a965838cf822a7c3",
          "content_type": "application/pdf",
          "doi": "10.1093/nargab/lqaf117",
          "id": "comprehensive-benchmark-of-differential-transcript-usa-final-published-version",
          "label": "Final published version",
          "publication_date": "2025-09-11",
          "retrieved_at": "2026-08-06T06:46:38+00:00",
          "source_access": "submitted-pdf",
          "status": "version-of-record"
        }
      ],
      "status": "published",
      "title": "Comprehensive benchmark of differential transcript usage analysis for bulk and single-cell RNA sequencing",
      "verification": {
        "last_verified": "2026-08-06",
        "notes": "AI-assisted double-pass extraction; production inclusion still requires the owner SHA-comment gate.",
        "status": "verified"
      },
      "work_type": "paper"
    },
    {
      "arxiv": null,
      "authors": [
        "Siyao Liu",
        "David L Corcoran",
        "Susana Garcia-Recio",
        "James S Marron",
        "Charles M Perou"
      ],
      "benchmark_ids": [
        "crafted-experiments"
      ],
      "benchmark_use_ids": [
        "crafted-experiments-to-evaluate-feature-selection-meth-crafted-experiments-1-use",
        "crafted-experiments-to-evaluate-feature-selection-meth-crafted-experiments-2-use"
      ],
      "canonical_url": "https://doi.org/10.1093/nargab/lqaf023",
      "current_version_id": "crafted-experiments-to-evaluate-feature-selection-meth-2025-03-19",
      "doi": "10.1093/nargab/lqaf023",
      "entity_type": "work",
      "evaluation_run_ids": [],
      "id": "crafted-experiments-to-evaluate-feature-selection-meth",
      "organizations": [
        "Lineberger Comprehensive Cancer Center, University of North Carolina",
        "University of North Carolina at Chapel Hill"
      ],
      "publication_date": "2025-03-19",
      "review_provenance": {
        "codex_cli_version": "codex-cli 0.146.0-alpha.9.2",
        "execution_surface": "local-codex-cli",
        "extractor_model_requested": "gpt-5.6-sol",
        "extractor_model_resolved": null,
        "generated_at": "2026-07-31T04:25:09+00:00",
        "local_run_id": "544b0fcb-9e98-44d1-9622-81b5ef6be124",
        "method": "local-codex-double-pass",
        "model_resolution_status": "not-reported",
        "pipeline_version": "1.4.0",
        "prompt_version": "paper-evidence-local-v15",
        "source_version_id": "crafted-experiments-to-evaluate-feature-selection-meth-2025-03-19",
        "verifier_model_requested": "gpt-5.6-sol",
        "verifier_model_resolved": null
      },
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": null,
          "canonical_url": "https://doi.org/10.1093/nargab/lqaf023",
          "content_sha256": "fec9ee9488a48bf24c0ec3062002fb1ec293ab627783e57889d5533d5d83420c",
          "content_type": "application/xml",
          "doi": "10.1093/nargab/lqaf023",
          "id": "crafted-experiments-to-evaluate-feature-selection-meth-2025-03-19",
          "label": "Reviewed source",
          "publication_date": "2025-03-19",
          "retrieved_at": "2026-07-31T04:25:09+00:00",
          "source_access": "open-url",
          "status": "version-of-record"
        }
      ],
      "status": "published",
      "title": "Crafted experiments to evaluate feature selection methods for single-cell RNA-seq data",
      "verification": {
        "last_verified": "2026-07-31",
        "notes": "AI-assisted double-pass extraction; production inclusion still requires the owner SHA-comment gate.",
        "status": "verified"
      },
      "work_type": "paper"
    },
    {
      "arxiv": null,
      "authors": [
        "Vishal Dey",
        "Xia Ning"
      ],
      "benchmark_ids": [
        "moleculenet"
      ],
      "benchmark_use_ids": [
        "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-1-use",
        "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-2-use",
        "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-3-use",
        "enhancing-molecular-property-prediction-with-auxiliary-moleculenet-4-use"
      ],
      "canonical_url": "https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11270959/fullTextXML",
      "current_version_id": "enhancing-molecular-property-prediction-with-auxiliary-pmc-version-1",
      "doi": "10.1186/s13321-024-00880-7",
      "entity_type": "work",
      "evaluation_run_ids": [],
      "id": "enhancing-molecular-property-prediction-with-auxiliary",
      "organizations": [
        "The Ohio State University"
      ],
      "publication_date": "2024-07-24",
      "review_provenance": {
        "codex_cli_version": "codex-cli 0.146.0-alpha.9.2",
        "execution_surface": "local-codex-cli",
        "extractor_model_requested": "gpt-5.6-sol",
        "extractor_model_resolved": null,
        "generated_at": "2026-08-02T04:01:46+00:00",
        "local_run_id": "09185fe4-a3fd-4558-a472-24b59dace0f4",
        "method": "local-codex-double-pass",
        "model_resolution_status": "not-reported",
        "pipeline_version": "1.4.0",
        "prompt_version": "paper-evidence-local-v15",
        "source_version_id": "enhancing-molecular-property-prediction-with-auxiliary-pmc-version-1",
        "verifier_model_requested": "gpt-5.6-sol",
        "verifier_model_resolved": null
      },
      "source_class": "independent_reproduction",
      "source_versions": [
        {
          "arxiv": null,
          "canonical_url": "https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11270959/fullTextXML",
          "content_sha256": "a7badc354868c05383f3df6991b2b45adf2b6ae7ab6f3b6a1c2cfc477c00e232",
          "content_type": "application/xml",
          "doi": "10.1186/s13321-024-00880-7",
          "id": "enhancing-molecular-property-prediction-with-auxiliary-pmc-version-1",
          "label": "PMC version 1",
          "publication_date": "2024-07-24",
          "retrieved_at": "2026-08-02T04:01:46+00:00",
          "source_access": "open-url",
          "status": "version-of-record"
        }
      ],
      "status": "published",
      "title": "Enhancing molecular property prediction with auxiliary learning and task-specific adaptation",
      "verification": {
        "last_verified": "2026-08-02",
        "notes": "AI-assisted double-pass extraction; production inclusion still requires the owner SHA-comment gate.",
        "status": "verified"
      },
      "work_type": "paper"
    },
    {
      "arxiv": null,
      "authors": [
        "Harshit Singh",
        "Rajeev Kumar Singh",
        "Satya Pratik Srivastava",
        "Suryavedha Pradhan",
        "Rohan Gorantla"
      ],
      "benchmark_ids": [
        "ab-bind",
        "abbibench",
        "ppb-affinity"
      ],
      "benchmark_use_ids": [
        "explainable-protein-protein-binding-affinity-predictio-ab-bind-5-use",
        "explainable-protein-protein-binding-affinity-predictio-ab-bind-6-use",
        "explainable-protein-protein-binding-affinity-predictio-ab-bind-7-use",
        "explainable-protein-protein-binding-affinity-predictio-ab-bind-8-use",
        "explainable-protein-protein-binding-affinity-predictio-abbibench-10-use",
        "explainable-protein-protein-binding-affinity-predictio-abbibench-11-use",
        "explainable-protein-protein-binding-affinity-predictio-abbibench-12-use",
        "explainable-protein-protein-binding-affinity-predictio-abbibench-9-use",
        "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-1-use",
        "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-2-use",
        "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-3-use",
        "explainable-protein-protein-binding-affinity-predictio-ppb-affinity-4-use"
      ],
      "canonical_url": "https://doi.org/10.64898/2026.03.30.715237",
      "current_version_id": "explainable-protein-protein-binding-affinity-predictio-biorxiv-version-posted-2",
      "doi": "10.64898/2026.03.30.715237",
      "entity_type": "work",
      "evaluation_run_ids": [],
      "id": "explainable-protein-protein-binding-affinity-predictio",
      "organizations": [
        "Department of Computer Science and Engineering, Shiv Nadar University, Delhi-NCR",
        "School of Informatics, University of Edinburgh",
        "EaStCHEM School of Chemistry, University of Edinburgh"
      ],
      "publication_date": "2026-04-01",
      "review_provenance": {
        "codex_cli_version": "codex-cli 0.148.0-alpha.9",
        "execution_surface": "local-codex-cli",
        "extractor_model_requested": "gpt-5.6-sol",
        "extractor_model_resolved": null,
        "generated_at": "2026-08-20T09:42:05+00:00",
        "local_run_id": "7aeadf4b-237c-4f01-b146-81fb3a885211",
        "method": "local-codex-double-pass",
        "model_resolution_status": "not-reported",
        "pipeline_version": "1.4.0",
        "prompt_version": "paper-evidence-local-v17",
        "source_version_id": "explainable-protein-protein-binding-affinity-predictio-biorxiv-version-posted-2",
        "verifier_model_requested": "gpt-5.6-sol",
        "verifier_model_resolved": null
      },
      "source_class": "independent_reproduction",
      "source_versions": [
        {
          "arxiv": null,
          "canonical_url": "https://doi.org/10.64898/2026.03.30.715237",
          "content_sha256": "a31273b2bdb98962327ea25dbf2e5a322fcbf1ba9edecebe87ceee7e20b0c1a7",
          "content_type": "application/pdf",
          "doi": "10.64898/2026.03.30.715237",
          "id": "explainable-protein-protein-binding-affinity-predictio-biorxiv-version-posted-2",
          "label": "bioRxiv version posted 2026-06-11",
          "publication_date": "2026-06-11",
          "retrieved_at": "2026-08-20T09:42:05+00:00",
          "source_access": "open-url",
          "status": "current"
        }
      ],
      "status": "preprint",
      "title": "Explainable protein-protein binding affinity prediction via fine-tuning protein language models",
      "verification": {
        "last_verified": "2026-08-20",
        "notes": "AI-assisted double-pass extraction; production inclusion still requires the owner SHA-comment gate.",
        "status": "verified"
      },
      "work_type": "preprint"
    },
    {
      "arxiv": null,
      "authors": [
        "Christian Dallago",
        "Jody Mou",
        "Kadina E. Johnston",
        "Bruce J. Wittmann",
        "Nicholas Bhattacharya",
        "Samuel Goldman",
        "Ali Madani",
        "Kevin K. Yang"
      ],
      "benchmark_ids": [
        "flip-aav",
        "flip-gb1",
        "flip-meltome"
      ],
      "benchmark_use_ids": [],
      "canonical_url": "https://openreview.net/forum?id=p2dMLEwL8tF",
      "current_version_id": "flip-paper-2021-10-11",
      "doi": "10.1101/2021.11.09.467890",
      "entity_type": "work",
      "evaluation_run_ids": [
        "flip-aav-des-mut",
        "flip-aav-low-vs-high",
        "flip-aav-mut-des",
        "flip-aav-one-vs-rest",
        "flip-aav-sampled",
        "flip-aav-seven-vs-rest",
        "flip-aav-two-vs-rest",
        "flip-gb1-low-vs-high",
        "flip-gb1-one-vs-rest",
        "flip-gb1-sampled",
        "flip-gb1-three-vs-rest",
        "flip-gb1-two-vs-rest",
        "flip-meltome-human",
        "flip-meltome-human-cell",
        "flip-meltome-mixed"
      ],
      "id": "flip-paper",
      "organizations": [
        "Technical University of Munich",
        "Microsoft Research New England",
        "California Institute of Technology",
        "University of California Berkeley",
        "Massachusetts Institute of Technology",
        "Salesforce Research"
      ],
      "publication_date": "2021-10-11",
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": null,
          "canonical_url": "https://openreview.net/forum?id=p2dMLEwL8tF",
          "doi": "10.1101/2021.11.09.467890",
          "id": "flip-paper-2021-10-11",
          "label": "Canonical source",
          "publication_date": "2021-10-11",
          "status": "current"
        }
      ],
      "status": "published",
      "title": "FLIP: Benchmark tasks in fitness landscape inference for proteins",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Published in the NeurIPS 2021 Datasets and Benchmarks Track; DOI identifies the creator-authorized bioRxiv version.",
        "status": "verified"
      },
      "work_type": "paper"
    },
    {
      "arxiv": null,
      "authors": [
        "Jeremiah H. Li",
        "Andrew J. Ho"
      ],
      "benchmark_ids": [
        "genebench-pro"
      ],
      "benchmark_use_ids": [],
      "canonical_url": "https://cdn.openai.com/pdf/21938268-21af-442f-af93-3b2249afb241/genebench-pro.pdf",
      "current_version_id": "genebench-pro-report-2026-06-30",
      "doi": "10.64898/2026.06.29.735386",
      "entity_type": "work",
      "evaluation_run_ids": [
        "genebench-pro-claude-high",
        "genebench-pro-claude-low",
        "genebench-pro-claude-max",
        "genebench-pro-claude-medium",
        "genebench-pro-claude-xhigh",
        "genebench-pro-official",
        "genebench-pro-pro-mode",
        "genebench-pro-reasoning-enabled",
        "genebench-pro-standard-high",
        "genebench-pro-standard-low",
        "genebench-pro-standard-max",
        "genebench-pro-standard-medium",
        "genebench-pro-standard-none"
      ],
      "id": "genebench-pro-report",
      "organizations": [
        "OpenAI"
      ],
      "publication_date": "2026-06-30",
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": null,
          "canonical_url": "https://cdn.openai.com/pdf/21938268-21af-442f-af93-3b2249afb241/genebench-pro.pdf",
          "doi": "10.64898/2026.06.29.735386",
          "id": "genebench-pro-report-2026-06-30",
          "label": "Canonical source",
          "publication_date": "2026-06-30",
          "status": "current"
        }
      ],
      "status": "published",
      "title": "GeneBench-Pro: Evaluating Multistage Statistical Reasoning in Genomics, Quantitative Biology, and Translational Biomedicine",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Official OpenAI PDF and bioRxiv v1 metadata; the PDF shortens the author names to Jeremy Li and Andrew Ho.",
        "status": "verified"
      },
      "work_type": "preprint"
    },
    {
      "arxiv": null,
      "authors": [
        "Katarína Grešová",
        "Vlastimil Martinek",
        "David Čechák",
        "Petr Šimeček",
        "Panagiotis Alexiou"
      ],
      "benchmark_ids": [
        "genomic-benchmarks"
      ],
      "benchmark_use_ids": [],
      "canonical_url": "https://doi.org/10.1186/s12863-023-01123-8",
      "current_version_id": "genomic-benchmarks-paper-2023-05-01",
      "doi": "10.1186/s12863-023-01123-8",
      "entity_type": "work",
      "evaluation_run_ids": [
        "genomic-benchmarks-creator-full"
      ],
      "id": "genomic-benchmarks-paper",
      "organizations": [
        "Masaryk University",
        "CEITEC Masaryk University"
      ],
      "publication_date": "2023-05-01",
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": null,
          "canonical_url": "https://doi.org/10.1186/s12863-023-01123-8",
          "doi": "10.1186/s12863-023-01123-8",
          "id": "genomic-benchmarks-paper-2023-05-01",
          "label": "Canonical source",
          "publication_date": "2023-05-01",
          "status": "current"
        }
      ],
      "status": "published",
      "title": "Genomic benchmarks: a collection of datasets for genomic sequence classification",
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Peer-reviewed open-access creator paper; Crossref metadata and publisher text verified.",
        "status": "verified"
      },
      "work_type": "paper"
    },
    {
      "arxiv": "1811.09621",
      "authors": [
        "Nathan Brown",
        "Marco Fiscato",
        "Marwin H. S. Segler",
        "Alain C. Vaucher"
      ],
      "benchmark_ids": [
        "guacamol"
      ],
      "benchmark_use_ids": [],
      "canonical_url": "https://doi.org/10.1021/acs.jcim.8b00839",
      "current_version_id": "guacamol-paper-2019-03-19",
      "doi": "10.1021/acs.jcim.8b00839",
      "entity_type": "work",
      "evaluation_run_ids": [
        "guacamol-creator-full"
      ],
      "id": "guacamol-paper",
      "organizations": [
        "BenevolentAI"
      ],
      "publication_date": "2019-03-19",
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": "1811.09621",
          "canonical_url": "https://doi.org/10.1021/acs.jcim.8b00839",
          "doi": "10.1021/acs.jcim.8b00839",
          "id": "guacamol-paper-2019-03-19",
          "label": "Canonical source",
          "publication_date": "2019-03-19",
          "status": "current"
        }
      ],
      "status": "published",
      "title": "GuacaMol: Benchmarking Models for de Novo Molecular Design",
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Peer-reviewed creator paper, official package, and official companion baselines verified.",
        "status": "verified"
      },
      "work_type": "paper"
    },
    {
      "arxiv": "2407.10362",
      "authors": [
        "Jon M. Laurent",
        "Joseph D. Janizek",
        "Michael Ruzo",
        "Michaela M. Hinks",
        "Michael J. Hammerling",
        "Siddharth Narayanan",
        "Manvitha Ponnapati",
        "Andrew D. White",
        "Samuel G. Rodriques"
      ],
      "benchmark_ids": [
        "lab-bench-cloning-scenarios",
        "lab-bench-dbqa-dga",
        "lab-bench-dbqa-gene-location",
        "lab-bench-dbqa-mirna-targets",
        "lab-bench-dbqa-mouse-tumor-gene-sets",
        "lab-bench-dbqa-oncogenic-signatures",
        "lab-bench-dbqa-tfbs-gtrd",
        "lab-bench-dbqa-variant-from-sequence",
        "lab-bench-dbqa-variant-multi-sequence",
        "lab-bench-dbqa-vax-response",
        "lab-bench-dbqa-viral-ppi",
        "lab-bench-figqa",
        "lab-bench-litqa2",
        "lab-bench-protocolqa",
        "lab-bench-seqqa-orf-seq-aaid",
        "lab-bench-seqqa-orf-seq-aaseq",
        "lab-bench-seqqa-orf-seq-numlen",
        "lab-bench-seqqa-orf-transeff",
        "lab-bench-seqqa-pcr-gene-enzprimers",
        "lab-bench-seqqa-pcr-gene-gibshindprimers",
        "lab-bench-seqqa-pcr-gene-gibssmaprimers",
        "lab-bench-seqqa-pcr-geneprimers-enz",
        "lab-bench-seqqa-pcr-len-primers",
        "lab-bench-seqqa-pcr-primers-len",
        "lab-bench-seqqa-pcr-seq-enzprimers",
        "lab-bench-seqqa-pcr-seq-primers",
        "lab-bench-seqqa-prop-seq-gcpercent",
        "lab-bench-seqqa-re-seq-lenfrags",
        "lab-bench-seqqa-re-seq-numfrags",
        "lab-bench-suppqa",
        "lab-bench-tableqa"
      ],
      "benchmark_use_ids": [],
      "canonical_url": "https://arxiv.org/abs/2407.10362v3",
      "current_version_id": "lab-bench-paper-2024-07-14",
      "doi": null,
      "entity_type": "work",
      "evaluation_run_ids": [
        "lab-bench-cloning-scenarios-creator-mcq",
        "lab-bench-cloning-scenarios-creator-mcq-llama-context",
        "lab-bench-cloning-scenarios-creator-open-response",
        "lab-bench-dbqa-dga-creator-mcq",
        "lab-bench-dbqa-gene-location-creator-mcq",
        "lab-bench-dbqa-mirna-targets-creator-mcq",
        "lab-bench-dbqa-mouse-tumor-gene-sets-creator-mcq",
        "lab-bench-dbqa-oncogenic-signatures-creator-mcq",
        "lab-bench-dbqa-tfbs-gtrd-creator-mcq",
        "lab-bench-dbqa-variant-from-sequence-creator-mcq",
        "lab-bench-dbqa-variant-multi-sequence-creator-mcq",
        "lab-bench-dbqa-vax-response-creator-mcq",
        "lab-bench-dbqa-viral-ppi-creator-mcq",
        "lab-bench-figqa-creator-mcq",
        "lab-bench-figqa-creator-open-response",
        "lab-bench-litqa2-creator-mcq",
        "lab-bench-protocolqa-creator-mcq",
        "lab-bench-protocolqa-creator-open-response",
        "lab-bench-seqqa-orf-seq-aaid-creator-mcq",
        "lab-bench-seqqa-orf-seq-aaseq-creator-mcq",
        "lab-bench-seqqa-orf-seq-numlen-creator-mcq",
        "lab-bench-seqqa-orf-transeff-creator-mcq",
        "lab-bench-seqqa-pcr-gene-enzprimers-creator-mcq",
        "lab-bench-seqqa-pcr-gene-gibshindprimers-creator-mcq",
        "lab-bench-seqqa-pcr-gene-gibssmaprimers-creator-mcq",
        "lab-bench-seqqa-pcr-geneprimers-enz-creator-mcq",
        "lab-bench-seqqa-pcr-len-primers-creator-mcq",
        "lab-bench-seqqa-pcr-primers-len-creator-mcq",
        "lab-bench-seqqa-pcr-seq-enzprimers-creator-mcq",
        "lab-bench-seqqa-pcr-seq-primers-creator-mcq",
        "lab-bench-seqqa-prop-seq-gcpercent-creator-mcq",
        "lab-bench-seqqa-re-seq-lenfrags-creator-mcq",
        "lab-bench-seqqa-re-seq-numfrags-creator-mcq",
        "lab-bench-suppqa-creator-mcq",
        "lab-bench-tableqa-creator-mcq"
      ],
      "id": "lab-bench-paper",
      "organizations": [
        "FutureHouse"
      ],
      "publication_date": "2024-07-14",
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": "2407.10362",
          "canonical_url": "https://arxiv.org/abs/2407.10362v3",
          "doi": null,
          "id": "lab-bench-paper-2024-07-14",
          "label": "Canonical source",
          "publication_date": "2024-07-14",
          "status": "current"
        }
      ],
      "status": "published",
      "title": "LAB-Bench: Measuring Capabilities of Language Models for Biology Research",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Creator preprint v3 and official repository; author list corrected to the canonical arXiv metadata.",
        "status": "verified"
      },
      "work_type": "paper"
    },
    {
      "arxiv": null,
      "authors": [
        "Amelia Liu",
        "Andrew Ho",
        "Anne Marie Droste",
        "David Martin",
        "Edmund Wong",
        "Edward Zhou",
        "Isabelle Zhou",
        "Joshua Park",
        "Joy Jiao",
        "Katie-Rose Skelly",
        "Kenny Kim",
        "Kevin Rao",
        "Masatoshi Uehara",
        "Max Marion",
        "Nicole Fitzgerald",
        "Rachel Dias",
        "Suyash Shringarpure",
        "Yuan Yuan",
        "Yunyun Wang"
      ],
      "benchmark_ids": [
        "lifescibench"
      ],
      "benchmark_use_ids": [],
      "canonical_url": "https://cdn.openai.com/pdf/b4299379-0a97-4ffa-8b9b-c3fbb299caa9/lifescibench_preprint.pdf",
      "current_version_id": "lifescibench-preprint-2026-06-17",
      "doi": null,
      "entity_type": "work",
      "evaluation_run_ids": [
        "lifescibench-official-full"
      ],
      "id": "lifescibench-preprint",
      "organizations": [
        "OpenAI",
        "Tacit Labs"
      ],
      "publication_date": "2026-06-17",
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": null,
          "canonical_url": "https://cdn.openai.com/pdf/b4299379-0a97-4ffa-8b9b-c3fbb299caa9/lifescibench_preprint.pdf",
          "doi": null,
          "id": "lifescibench-preprint-2026-06-17",
          "label": "Canonical source",
          "publication_date": "2026-06-17",
          "status": "current"
        }
      ],
      "status": "preprint",
      "title": "LifeSciBench: Evaluating Language Models on Realistic, Expert-Level Tasks in the Life Sciences",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Title, author list, and affiliations transcribed from the official OpenAI-hosted preprint; publication date follows the official release page.",
        "status": "verified"
      },
      "work_type": "preprint"
    },
    {
      "arxiv": "1703.00564",
      "authors": [
        "Zhenqin Wu",
        "Bharath Ramsundar",
        "Evan N. Feinberg",
        "Joseph Gomes",
        "Caleb Geniesse",
        "Aneesh S. Pappu",
        "Karl Leswing",
        "Vijay Pande"
      ],
      "benchmark_ids": [
        "moleculenet"
      ],
      "benchmark_use_ids": [],
      "canonical_url": "https://doi.org/10.1039/C7SC02664A",
      "current_version_id": "moleculenet-paper-2017-10-31",
      "doi": "10.1039/C7SC02664A",
      "entity_type": "work",
      "evaluation_run_ids": [
        "moleculenet-creator-full"
      ],
      "id": "moleculenet-paper",
      "organizations": [
        "Stanford University",
        "DeepChem"
      ],
      "publication_date": "2017-10-31",
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": "1703.00564",
          "canonical_url": "https://doi.org/10.1039/C7SC02664A",
          "doi": "10.1039/C7SC02664A",
          "id": "moleculenet-paper-2017-10-31",
          "label": "Canonical source",
          "publication_date": "2017-10-31",
          "status": "current"
        }
      ],
      "status": "published",
      "title": "MoleculeNet: a benchmark for molecular machine learning",
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Peer-reviewed Chemical Science creator paper and open DeepChem implementation verified.",
        "status": "verified"
      },
      "work_type": "paper"
    },
    {
      "arxiv": null,
      "authors": [
        "Huaqing Liu",
        "Peiyi Chen",
        "Xiaochen Zhai",
        "Ku-Geng Huo",
        "Shuxian Zhou",
        "Lanqing Han",
        "Guoxin Fan"
      ],
      "benchmark_ids": [
        "ppb-affinity"
      ],
      "benchmark_use_ids": [
        "ppb-affinity-protein-protein-binding-affinity-dataset-ppb-affinity-1-use",
        "ppb-affinity-protein-protein-binding-affinity-dataset-ppb-affinity-2-use",
        "ppb-affinity-protein-protein-binding-affinity-dataset-ppb-affinity-3-use"
      ],
      "canonical_url": "https://doi.org/10.1038/s41597-024-03997-4",
      "current_version_id": "ppb-affinity-protein-protein-binding-affinity-dataset-2024-12-03",
      "doi": "10.1038/s41597-024-03997-4",
      "entity_type": "work",
      "evaluation_run_ids": [],
      "id": "ppb-affinity-protein-protein-binding-affinity-dataset",
      "organizations": [
        "Artificial Intelligence Innovation Center, Research Institute of Tsinghua, Pearl River Delta",
        "Cyagen Biosciences (Suzhou) Inc.",
        "Cyagen Biosciences (Guangzhou) Inc.",
        "Cyagen Biomodels (Guangzhou) Co., Ltd",
        "Department of Pain Medicine, Shenzhen Nanshan People’s Hospital, Shenzhen University Medical School"
      ],
      "publication_date": "2024-12-03",
      "review_provenance": {
        "codex_cli_version": "codex-cli 0.146.0-alpha.3.1",
        "execution_surface": "local-codex-cli",
        "extractor_model_requested": "gpt-5.6-sol",
        "extractor_model_resolved": null,
        "generated_at": "2026-07-29T23:16:51+00:00",
        "local_run_id": "6093aa15-7a9b-4fb8-80f4-bb8eae2c9e14",
        "method": "local-codex-double-pass",
        "model_resolution_status": "not-reported",
        "pipeline_version": "1.4.0",
        "prompt_version": "paper-evidence-local-v10",
        "source_version_id": "ppb-affinity-protein-protein-binding-affinity-dataset-2024-12-03",
        "verifier_model_requested": "gpt-5.6-sol",
        "verifier_model_resolved": null
      },
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": null,
          "canonical_url": "https://doi.org/10.1038/s41597-024-03997-4",
          "content_sha256": "f3bcd9a3282a7c906ce75c3b847ba24f27ecbf9ce876e9c790d710f5e676b35a",
          "content_type": "application/xml",
          "doi": "10.1038/s41597-024-03997-4",
          "id": "ppb-affinity-protein-protein-binding-affinity-dataset-2024-12-03",
          "label": "Reviewed source",
          "publication_date": "2024-12-03",
          "retrieved_at": "2026-07-29T23:16:51+00:00",
          "source_access": "submitted-pdf",
          "status": "version-of-record"
        }
      ],
      "status": "published",
      "title": "PPB-Affinity: Protein-Protein Binding Affinity dataset for AI-based protein drug discovery",
      "verification": {
        "last_verified": "2026-07-29",
        "notes": "AI-assisted double-pass extraction; production inclusion still requires the owner SHA-comment gate.",
        "status": "verified"
      },
      "work_type": "paper"
    },
    {
      "arxiv": null,
      "authors": [
        "Pascal Notin",
        "Aaron W. Kollasch",
        "Daniel Ritter",
        "Lood van Niekerk",
        "Steffanie Paul",
        "Hansen Spinner",
        "Nathan Rollins",
        "Ada Shaw",
        "Ruben Weitzman",
        "Jonathan Frazer",
        "Mafalda Dias",
        "Dinko Franceschi",
        "Rose Orenbuch",
        "Yarin Gal",
        "Debora S. Marks"
      ],
      "benchmark_ids": [
        "proteingym-dms-substitutions"
      ],
      "benchmark_use_ids": [],
      "canonical_url": "https://papers.nips.cc/paper_files/paper/2023/hash/cac723e5ff29f65e3fcbb0739ae91bee-Abstract-Datasets_and_Benchmarks.html",
      "current_version_id": "proteingym-paper-2023-12-10",
      "doi": null,
      "entity_type": "work",
      "evaluation_run_ids": [
        "proteingym-v10-dms-substitutions-zero-shot"
      ],
      "id": "proteingym-paper",
      "organizations": [
        "University of Oxford",
        "Harvard Medical School",
        "Seismic Therapeutic",
        "Harvard University",
        "Centre for Genomic Regulation",
        "Universitat Pompeu Fabra",
        "Broad Institute"
      ],
      "publication_date": "2023-12-10",
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": null,
          "canonical_url": "https://papers.nips.cc/paper_files/paper/2023/hash/cac723e5ff29f65e3fcbb0739ae91bee-Abstract-Datasets_and_Benchmarks.html",
          "doi": null,
          "id": "proteingym-paper-2023-12-10",
          "label": "Canonical source",
          "publication_date": "2023-12-10",
          "status": "current"
        }
      ],
      "status": "published",
      "title": "ProteinGym: Large-Scale Benchmarks for Protein Fitness Prediction and Design",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Final NeurIPS 2023 Datasets and Benchmarks proceedings paper. The official PDF spells the sixth author Hansen Spinner, while the NeurIPS program/BibTeX says Han Spinner; this record follows the final PDF author line. Publication date is normalized to the first day of NeurIPS 2023.",
        "status": "verified"
      },
      "work_type": "paper"
    },
    {
      "arxiv": "2406.05540",
      "authors": [
        "Yiqing Shen",
        "Zan Chen",
        "Michail Mamalakis",
        "Luhan He",
        "Haiyang Xia",
        "Tianbin Li",
        "Yanzhou Su",
        "Junjun He",
        "Yu Guang Wang"
      ],
      "benchmark_ids": [
        "proteinlmbench"
      ],
      "benchmark_use_ids": [],
      "canonical_url": "https://arxiv.org/abs/2406.05540",
      "current_version_id": "proteinlmbench-paper-2024-06-08",
      "doi": null,
      "entity_type": "work",
      "evaluation_run_ids": [
        "proteinlmbench-creator-full"
      ],
      "id": "proteinlmbench-paper",
      "organizations": [
        "Toursun Synbio",
        "Johns Hopkins University",
        "University of Cambridge",
        "Shanghai Institute for Biomedical and Pharmaceutical Technologies",
        "Shanghai AI Laboratory",
        "Shanghai Jiao Tong University",
        "UNSW Sydney"
      ],
      "publication_date": "2024-06-08",
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": "2406.05540",
          "canonical_url": "https://arxiv.org/abs/2406.05540",
          "doi": null,
          "id": "proteinlmbench-paper-2024-06-08",
          "label": "Canonical source",
          "publication_date": "2024-06-08",
          "status": "current"
        }
      ],
      "status": "preprint",
      "title": "A Fine-tuning Dataset and Benchmark for Large Language Models for Protein Understanding",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Creator preprint v2 dated 2024-07-08; first arXiv submission was 2024-06-08.",
        "status": "verified"
      },
      "work_type": "preprint"
    },
    {
      "arxiv": "2602.09063",
      "authors": [
        "Kenny Workman",
        "Zhen Yang",
        "Harihara Muralidharan",
        "Aidan Abdulali",
        "Hannah Le"
      ],
      "benchmark_ids": [
        "scbench"
      ],
      "benchmark_use_ids": [
        "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an-scbench-1-use",
        "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an-scbench-2-use"
      ],
      "canonical_url": "https://arxiv.org/abs/2602.09063",
      "current_version_id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an-arxiv-v1",
      "doi": null,
      "entity_type": "work",
      "evaluation_run_ids": [],
      "id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an",
      "organizations": [
        "LatchBio"
      ],
      "publication_date": "2026-02-09",
      "review_provenance": {
        "codex_cli_version": "codex-cli 0.146.0-alpha.3.1",
        "execution_surface": "local-codex-cli",
        "extractor_model_requested": "gpt-5.6-sol",
        "extractor_model_resolved": null,
        "generated_at": "2026-07-27T21:50:16+00:00",
        "local_run_id": "f647b32e-2194-47e1-a6b7-5a4ea21cb20c",
        "method": "local-codex-double-pass",
        "model_resolution_status": "not-reported",
        "pipeline_version": "1.4.0",
        "prompt_version": "paper-evidence-local-v7",
        "source_version_id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an-arxiv-v1",
        "verifier_model_requested": "gpt-5.6-sol",
        "verifier_model_resolved": null
      },
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": "2602.09063",
          "canonical_url": "https://arxiv.org/abs/2602.09063",
          "content_sha256": "6cebd7ca535d38c0f93fc1b7daa8f5262ebc3868d1da953cb5249ccb5df3223d",
          "content_type": "application/pdf",
          "doi": null,
          "id": "scbench-evaluating-ai-agents-on-single-cell-rna-seq-an-arxiv-v1",
          "label": "arXiv v1",
          "publication_date": "2026-02-09",
          "retrieved_at": "2026-07-27T21:50:16+00:00",
          "source_access": "submitted-pdf",
          "status": "current"
        }
      ],
      "status": "preprint",
      "title": "scBench: Evaluating AI Agents on Single-Cell RNA-seq Analysis",
      "verification": {
        "last_verified": "2026-07-27",
        "notes": "AI-assisted double-pass extraction; production inclusion still requires the owner SHA-comment gate.",
        "status": "verified"
      },
      "work_type": "preprint"
    },
    {
      "arxiv": null,
      "authors": [
        "Malte D. Luecken",
        "M. Büttner",
        "K. Chaichoompu",
        "A. Danese",
        "M. Interlandi",
        "M. F. Mueller",
        "D. C. Strobl",
        "L. Zappia",
        "M. Dugas",
        "M. Colomé-Tatché",
        "Fabian J. Theis"
      ],
      "benchmark_ids": [
        "scib"
      ],
      "benchmark_use_ids": [],
      "canonical_url": "https://doi.org/10.1038/s41592-021-01336-8",
      "current_version_id": "scib-paper-2021-12-23",
      "doi": "10.1038/s41592-021-01336-8",
      "entity_type": "work",
      "evaluation_run_ids": [
        "scib-creator-full"
      ],
      "id": "scib-paper",
      "organizations": [
        "Helmholtz Zentrum München",
        "Technical University of Munich"
      ],
      "publication_date": "2021-12-23",
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": null,
          "canonical_url": "https://doi.org/10.1038/s41592-021-01336-8",
          "doi": "10.1038/s41592-021-01336-8",
          "id": "scib-paper-2021-12-23",
          "label": "Canonical source",
          "publication_date": "2021-12-23",
          "status": "current"
        }
      ],
      "status": "published",
      "title": "Benchmarking atlas-level data integration in single-cell genomics",
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Peer-reviewed Nature Methods creator paper and paper-specific reproducibility pipeline verified.",
        "status": "verified"
      },
      "work_type": "paper"
    },
    {
      "arxiv": "2507.02083",
      "authors": [
        "Haonan Duan",
        "Stephen Zhewen Lu",
        "Caitlin F. Harrigan",
        "Nishkrit Desai",
        "Jiarui Lu",
        "Michał Koziarski",
        "Leonardo Cotta",
        "Chris J. Maddison"
      ],
      "benchmark_ids": [
        "scigym-small"
      ],
      "benchmark_use_ids": [],
      "canonical_url": "https://proceedings.neurips.cc/paper_files/paper/2025/hash/a23760ba51036d530a2656c3835f826c-Abstract-Datasets_and_Benchmarks_Track.html",
      "current_version_id": "scigym-paper-2025-11-30",
      "doi": null,
      "entity_type": "work",
      "evaluation_run_ids": [
        "scigym-small-creator-paper",
        "scigym-small-zero-shot"
      ],
      "id": "scigym-paper",
      "organizations": [
        "University of Toronto",
        "SickKids",
        "Axiom",
        "Mila",
        "Vector Institute"
      ],
      "publication_date": "2025-11-30",
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": "2507.02083",
          "canonical_url": "https://proceedings.neurips.cc/paper_files/paper/2025/hash/a23760ba51036d530a2656c3835f826c-Abstract-Datasets_and_Benchmarks_Track.html",
          "doi": null,
          "id": "scigym-paper-2025-11-30",
          "label": "Canonical source",
          "publication_date": "2025-11-30",
          "status": "current"
        }
      ],
      "status": "published",
      "title": "Measuring Scientific Capabilities of Language Models with a Systems Biology Dry Lab",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Final Advances in Neural Information Processing Systems 38 Datasets and Benchmarks Track paper. The publication date is normalized to the first official NeurIPS 2025 meeting date; the benchmark was first publicly released on 2025-05-16.",
        "status": "verified"
      },
      "work_type": "paper"
    },
    {
      "arxiv": null,
      "authors": [
        "Junhao Liu",
        "Siwei Xu",
        "Yongxian Wu",
        "Jing Zhang"
      ],
      "benchmark_ids": [
        "single-cell-omics-arena-soar"
      ],
      "benchmark_use_ids": [
        "single-cell-omics-arena-evaluation-of-large-language-m-single-cell-omics-arena-soar-1-use"
      ],
      "canonical_url": "https://doi.org/10.1093/bib/bbaf622",
      "current_version_id": "single-cell-omics-arena-evaluation-of-large-language-m-pmc-version-1",
      "doi": "10.1093/bib/bbaf622",
      "entity_type": "work",
      "evaluation_run_ids": [],
      "id": "single-cell-omics-arena-evaluation-of-large-language-m",
      "organizations": [
        "University of California, Irvine"
      ],
      "publication_date": "2025-11-24",
      "review_provenance": {
        "codex_cli_version": "codex-cli 0.147.0-alpha.6.5",
        "execution_surface": "local-codex-cli",
        "extractor_model_requested": "gpt-5.6-sol",
        "extractor_model_resolved": null,
        "generated_at": "2026-08-13T04:19:47+00:00",
        "local_run_id": "8dce10cb-b039-4b4a-a89a-997ed04bdbf0",
        "method": "local-codex-double-pass",
        "model_resolution_status": "not-reported",
        "pipeline_version": "1.4.0",
        "prompt_version": "paper-evidence-local-v17",
        "source_version_id": "single-cell-omics-arena-evaluation-of-large-language-m-pmc-version-1",
        "verifier_model_requested": "gpt-5.6-sol",
        "verifier_model_resolved": null
      },
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": null,
          "canonical_url": "https://doi.org/10.1093/bib/bbaf622",
          "content_sha256": "5057fff087ca40a60c91f9eb9644b43ff02426f9b004b389cbdf006b3394394e",
          "content_type": "application/xml",
          "doi": "10.1093/bib/bbaf622",
          "id": "single-cell-omics-arena-evaluation-of-large-language-m-pmc-version-1",
          "label": "PMC version 1",
          "publication_date": "2025-11-24",
          "retrieved_at": "2026-08-13T04:19:47+00:00",
          "source_access": "open-url",
          "status": "version-of-record"
        }
      ],
      "status": "published",
      "title": "Single-cell omics arena: evaluation of large language models for automatic cell-type annotations on single-cell omics data via RNA-seq bridging",
      "verification": {
        "last_verified": "2026-08-13",
        "notes": "AI-assisted double-pass extraction; production inclusion still requires the owner SHA-comment gate.",
        "status": "verified"
      },
      "work_type": "paper"
    },
    {
      "arxiv": null,
      "authors": [
        "Junhao Liu"
      ],
      "benchmark_ids": [
        "single-cell-omics-arena-soar"
      ],
      "benchmark_use_ids": [
        "soar-e5d2b3e-rna-zero-shot-cot-use",
        "soar-e5d2b3e-rna-zero-shot-use"
      ],
      "canonical_url": "https://github.com/aicb-ZhangLabs/SOAR/tree/e5d2b3e2619cb56fece5fba78fae989a67fd0c13",
      "current_version_id": "single-cell-omics-arena-soar-repository-result-snapshot-e5d2b3e",
      "doi": null,
      "entity_type": "work",
      "evaluation_run_ids": [
        "soar-e5d2b3e-rna-zero-shot",
        "soar-e5d2b3e-rna-zero-shot-cot"
      ],
      "id": "single-cell-omics-arena-soar-repository-result-snapshot",
      "organizations": [
        "University of California, Irvine"
      ],
      "publication_date": "2025-08-03",
      "review_provenance": {
        "codex_cli_version": "codex-cli 0.147.0-alpha.6.5",
        "execution_surface": "local-codex-cli",
        "extractor_model_requested": "gpt-5.6-sol",
        "extractor_model_resolved": null,
        "generated_at": "2026-08-13T06:19:00Z",
        "local_run_id": "7ec178bf-2423-4937-9441-7f31f7f21f24",
        "method": "local-codex-double-pass",
        "model_resolution_status": "not-reported",
        "pipeline_version": "1.4.0",
        "prompt_version": "paper-evidence-local-v17",
        "source_version_id": "single-cell-omics-arena-soar-repository-result-snapshot-e5d2b3e",
        "verifier_model_requested": "gpt-5.6-sol",
        "verifier_model_resolved": null
      },
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": null,
          "canonical_url": "https://github.com/aicb-ZhangLabs/SOAR/tree/e5d2b3e2619cb56fece5fba78fae989a67fd0c13",
          "content_sha256": "3e2e46418c496797996f3a00ad6a525fa3606dac8908c062f6c4028d5d21e5ab",
          "content_type": "text/plain",
          "doi": null,
          "id": "single-cell-omics-arena-soar-repository-result-snapshot-e5d2b3e",
          "label": "Repository snapshot e5d2b3e",
          "publication_date": "2025-08-03",
          "retrieved_at": "2026-08-13T06:19:00Z",
          "source_access": "open-url",
          "status": "current"
        }
      ],
      "status": "published",
      "title": "Single-cell Omics Arena repository result snapshot",
      "verification": {
        "last_verified": "2026-08-13",
        "notes": "Commit-pinned official result snapshot; result and protocol claims passed independent local double-pass verification.",
        "status": "verified"
      },
      "work_type": "official-release"
    },
    {
      "arxiv": "2512.21907",
      "authors": [
        "Kenny Workman",
        "Zhen Yang",
        "Harihara Muralidharan",
        "Hannah Le"
      ],
      "benchmark_ids": [
        "spatialbench"
      ],
      "benchmark_use_ids": [
        "spatialbench-preprint-creation",
        "spatialbench-preprint-evaluation"
      ],
      "canonical_url": "https://arxiv.org/abs/2512.21907",
      "current_version_id": "spatialbench-preprint-v2",
      "doi": "10.48550/arXiv.2512.21907",
      "entity_type": "work",
      "evaluation_run_ids": [
        "spatialbench-paper-v2-base",
        "spatialbench-paper-v2-claude-code",
        "spatialbench-paper-v2-latch"
      ],
      "id": "spatialbench-preprint",
      "organizations": [
        "LatchBio"
      ],
      "publication_date": "2025-12-26",
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": "2512.21907v1",
          "canonical_url": "https://arxiv.org/abs/2512.21907v1",
          "doi": "10.48550/arXiv.2512.21907",
          "id": "spatialbench-preprint-v1",
          "label": "arXiv v1",
          "publication_date": "2025-12-26",
          "status": "superseded"
        },
        {
          "arxiv": "2512.21907v2",
          "canonical_url": "https://arxiv.org/abs/2512.21907v2",
          "doi": "10.48550/arXiv.2512.21907",
          "id": "spatialbench-preprint-v2",
          "label": "arXiv v2",
          "publication_date": "2026-01-05",
          "status": "current"
        }
      ],
      "status": "preprint",
      "title": "SpatialBench: Can Agents Analyze Real-World Spatial Biology Data?",
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "arXiv identity, v1/v2 dates, 146-problem inventory, methods, and printed tables were verified against version 2.",
        "status": "verified"
      },
      "work_type": "preprint"
    },
    {
      "arxiv": null,
      "authors": [
        "LatchBio"
      ],
      "benchmark_ids": [
        "spatialbench"
      ],
      "benchmark_use_ids": [
        "spatialbench-repository-creation",
        "spatialbench-repository-evaluation"
      ],
      "canonical_url": "https://github.com/latchbio/spatialbench/tree/5042c4f3ee597da1590650c7b894d068ae968e26",
      "current_version_id": "spatialbench-repository-release-2026-06-10",
      "doi": null,
      "entity_type": "work",
      "evaluation_run_ids": [
        "spatialbench-repo-159-claude-code",
        "spatialbench-repo-159-mini-swe-agent",
        "spatialbench-repo-159-openai-codex",
        "spatialbench-repo-159-pi"
      ],
      "id": "spatialbench-repository-release",
      "organizations": [
        "LatchBio"
      ],
      "publication_date": "2026-06-10",
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": null,
          "canonical_url": "https://github.com/latchbio/spatialbench/tree/5042c4f3ee597da1590650c7b894d068ae968e26",
          "doi": null,
          "id": "spatialbench-repository-release-2026-06-10",
          "label": "Canonical source",
          "publication_date": "2026-06-10",
          "status": "current"
        }
      ],
      "status": "published",
      "title": "SpatialBench 159-evaluation repository snapshot",
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "Creator-maintained commit-pinned snapshot with 159-evaluation metadata, methods, graders, trajectories, and exact result tables.",
        "status": "verified"
      },
      "work_type": "official-release"
    },
    {
      "arxiv": null,
      "authors": [],
      "benchmark_ids": [
        "biomysterybench",
        "proteingym",
        "scbench",
        "spatialbench"
      ],
      "benchmark_use_ids": [
        "system-card-claude-opus-5-biomysterybench-1-use",
        "system-card-claude-opus-5-proteingym-4-use",
        "system-card-claude-opus-5-scbench-3-use",
        "system-card-claude-opus-5-spatialbench-2-use"
      ],
      "canonical_url": "https://anthropic.com/",
      "current_version_id": "system-card-claude-opus-5-2026-07-24",
      "doi": null,
      "entity_type": "work",
      "evaluation_run_ids": [],
      "id": "system-card-claude-opus-5",
      "organizations": [
        "Anthropic"
      ],
      "publication_date": "2026-07-24",
      "review_provenance": {
        "codex_cli_version": "codex-cli 0.146.0-alpha.3.1",
        "execution_surface": "local-codex-cli",
        "extractor_model_requested": "gpt-5.6-sol",
        "extractor_model_resolved": null,
        "generated_at": "2026-07-29T01:35:14+00:00",
        "local_run_id": "02010851-fccb-4a95-aa0e-663a39afbfea",
        "method": "local-codex-double-pass",
        "model_resolution_status": "not-reported",
        "pipeline_version": "1.4.0",
        "prompt_version": "paper-evidence-local-v8",
        "source_version_id": "system-card-claude-opus-5-2026-07-24",
        "verifier_model_requested": "gpt-5.6-sol",
        "verifier_model_resolved": null
      },
      "source_class": "official_model_provider",
      "source_versions": [
        {
          "arxiv": null,
          "canonical_url": "https://anthropic.com/",
          "content_sha256": "897768f0f6f1724f3109279ab3f6458c9fbf496b56d5d2be14cab3a4f91ca472",
          "content_type": "application/pdf",
          "doi": null,
          "id": "system-card-claude-opus-5-2026-07-24",
          "label": "Reviewed source",
          "publication_date": "2026-07-24",
          "retrieved_at": "2026-07-29T01:35:14+00:00",
          "source_access": "submitted-pdf",
          "status": "current"
        }
      ],
      "status": "published",
      "title": "System Card: Claude Opus 5",
      "verification": {
        "last_verified": "2026-07-29",
        "notes": "AI-assisted double-pass extraction; production inclusion still requires the owner SHA-comment gate.",
        "status": "verified"
      },
      "work_type": "system-card"
    },
    {
      "arxiv": "1906.08230",
      "authors": [
        "Roshan Rao",
        "Nicholas Bhattacharya",
        "Neil Thomas",
        "Yan Duan",
        "Xi Chen",
        "John Canny",
        "Pieter Abbeel",
        "Yun S. Song"
      ],
      "benchmark_ids": [
        "tape"
      ],
      "benchmark_use_ids": [],
      "canonical_url": "https://proceedings.neurips.cc/paper/2019/hash/37f65c068b7723cd7809ee2d31d7861c-Abstract.html",
      "current_version_id": "tape-paper-2019-12-08",
      "doi": "10.1101/676825",
      "entity_type": "work",
      "evaluation_run_ids": [
        "tape-creator-full"
      ],
      "id": "tape-paper",
      "organizations": [
        "University of California Berkeley"
      ],
      "publication_date": "2019-12-08",
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": "1906.08230",
          "canonical_url": "https://proceedings.neurips.cc/paper/2019/hash/37f65c068b7723cd7809ee2d31d7861c-Abstract.html",
          "doi": "10.1101/676825",
          "id": "tape-paper-2019-12-08",
          "label": "Canonical source",
          "publication_date": "2019-12-08",
          "status": "current"
        }
      ],
      "status": "published",
      "title": "Evaluating Protein Transfer Learning with TAPE",
      "verification": {
        "last_verified": "2026-07-22",
        "notes": "NeurIPS 2019 creator paper with public data and code.",
        "status": "verified"
      },
      "work_type": "paper"
    },
    {
      "arxiv": null,
      "authors": [
        "Anthropic"
      ],
      "benchmark_ids": [
        "virbench"
      ],
      "benchmark_use_ids": [],
      "canonical_url": "https://www.anthropic.com/research/agents-in-biology",
      "current_version_id": "virbench-official-2025-05-20",
      "doi": null,
      "entity_type": "work",
      "evaluation_run_ids": [
        "virbench-official-run"
      ],
      "id": "virbench-official",
      "organizations": [
        "Anthropic"
      ],
      "publication_date": "2025-05-20",
      "source_class": "benchmark_creator",
      "source_versions": [
        {
          "arxiv": null,
          "canonical_url": "https://www.anthropic.com/research/agents-in-biology",
          "doi": null,
          "id": "virbench-official-2025-05-20",
          "label": "Canonical source",
          "publication_date": "2025-05-20",
          "status": "current"
        }
      ],
      "status": "published",
      "title": "Paving the way for AI agents in biology",
      "verification": {
        "last_verified": "2026-07-21",
        "notes": "Official Anthropic research page introducing VirBench.",
        "status": "verified"
      },
      "work_type": "official-release"
    }
  ]
}
